Add tests and update README
This commit is contained in:
parent
056f93e355
commit
072ca8b0ed
8 changed files with 202 additions and 23 deletions
17
Makefile
17
Makefile
|
|
@ -1,6 +1,6 @@
|
|||
SHELL := bash
|
||||
|
||||
.PHONY: check clean reformat dist docker amd64 index
|
||||
.PHONY: check clean reformat dist test
|
||||
|
||||
all: dist
|
||||
|
||||
|
|
@ -13,17 +13,8 @@ check:
|
|||
reformat:
|
||||
scripts/format-code.sh
|
||||
|
||||
test:
|
||||
scripts/run-tests.sh
|
||||
|
||||
dist:
|
||||
python3 setup.py sdist
|
||||
|
||||
docker:
|
||||
scripts/build-docker.sh
|
||||
for lang in de-de en-us es-es fr-fr it-it nl ru-ru sv-se; do \
|
||||
LARYNX_LANGUAGE=$$lang scripts/build-docker.sh; \
|
||||
done
|
||||
|
||||
amd64:
|
||||
NOBUILDX=1 scripts/build-docker.sh
|
||||
|
||||
index:
|
||||
bin/make_sample_html.py local/ > index.html
|
||||
|
|
|
|||
75
README.md
75
README.md
|
|
@ -1 +1,76 @@
|
|||
# eSpeak Phonemizer
|
||||
|
||||
Uses [ctypes](https://docs.python.org/3/library/ctypes.html) and [libespeak-ng](https://github.com/espeak-ng/espeak-ng/blob/master/docs/integration.md) to transform text into [IPA](https://en.wikipedia.org/wiki/International_Phonetic_Alphabet) phonemes.
|
||||
|
||||
## Installation
|
||||
|
||||
First, install libespeak-ng:
|
||||
|
||||
```sh
|
||||
sudo apt-get install libespeak-ng1
|
||||
```
|
||||
|
||||
Next, install espeak_phonemizer:
|
||||
|
||||
```sh
|
||||
pip install espeak_phonemizer
|
||||
```
|
||||
|
||||
If installation was successful, you should be able to run:
|
||||
|
||||
```sh
|
||||
espeak-phonemizer --version
|
||||
```
|
||||
|
||||
## Basic Phonemization
|
||||
|
||||
Simply pass your text into the standard input of `espeak-phonemizer`:
|
||||
|
||||
```sh
|
||||
echo 'This is a test.' | espeak-phonemizer -v en-us
|
||||
ðɪs ɪz ɐ tˈɛst
|
||||
```
|
||||
|
||||
### Separators
|
||||
|
||||
Phoneme and word separators can be changed:
|
||||
|
||||
```sh
|
||||
echo 'This is a test.' | espeak-phonemizer -v en-us -p '_' -w '#'
|
||||
ð_ɪ_s#ɪ_z#ɐ#t_ˈɛ_s_t
|
||||
```
|
||||
|
||||
### Punctuation and Stress
|
||||
|
||||
Some punctuation can be kept (.,;:!?) in the output:
|
||||
|
||||
```sh
|
||||
echo 'This: is, a, test.' | espeak-phonemizer -v en-us --keep-punctuation
|
||||
ðˈɪs: ˈɪz, ˈeɪ, tˈɛst.
|
||||
```
|
||||
|
||||
Stress markers can also be dropped:
|
||||
|
||||
```sh
|
||||
echo 'This is a test.' | espeak-phonemizer -v en-us --no-stress
|
||||
ðɪs ɪz ɐ tɛst
|
||||
```
|
||||
|
||||
### Delimited Input
|
||||
|
||||
The `--csv` flag enables delimited input with fields separated by a '|' (change with `--csv-delimiter`):
|
||||
|
||||
```sh
|
||||
echo 's1|This is a test.' | espeak-phonemizer -v en-us --csv
|
||||
s1|This is a test.|ðɪs ɪz ɐ tˈɛst
|
||||
```
|
||||
|
||||
Phonemes are added as a final column, allowing you to pass arbitrary metadata through to the output.
|
||||
|
||||
### Parallelize with GNU Parallel
|
||||
|
||||
```sh
|
||||
parallel -a /path/to/input.csv --pipepart \
|
||||
espeak-phonemizer -v en-us --csv \
|
||||
> /path/to/output.csv
|
||||
```
|
||||
|
|
|
|||
|
|
@ -1 +1 @@
|
|||
1.0
|
||||
1.0.0
|
||||
|
|
@ -1,10 +1,23 @@
|
|||
"""Uses ctypes and libespeak-ng to get IPA phonemes from text"""
|
||||
import ctypes
|
||||
import re
|
||||
import typing
|
||||
from pathlib import Path
|
||||
|
||||
_DIR = Path(__file__).parent
|
||||
__version__ = (_DIR / "VERSION").read_text().strip()
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
||||
class Phonemizer:
|
||||
"""Use ctypes and libespeak-ng to get IPA phonemes from text"""
|
||||
"""
|
||||
Use ctypes and libespeak-ng to get IPA phonemes from text.
|
||||
Not thread safe.
|
||||
|
||||
Requires libc.so.6
|
||||
Tries to use libespeak-ng.so or libespeak-ng.so.1
|
||||
"""
|
||||
|
||||
SEEK_SET = 0
|
||||
|
||||
|
|
@ -44,7 +57,23 @@ class Phonemizer:
|
|||
keep_language_flags: bool = False,
|
||||
no_stress: bool = False,
|
||||
) -> str:
|
||||
"""Return IPA string for text"""
|
||||
"""
|
||||
Return IPA string for text.
|
||||
Not thread safe.
|
||||
|
||||
Args:
|
||||
text: Text to phonemize
|
||||
voice: optional voice (uses self.default_voice if None)
|
||||
keep_clause_breakers: True if punctuation symbols should be kept
|
||||
phoneme_separator: Separator character between phonemes
|
||||
word_separator: Separator string between words (default: space)
|
||||
punctuation_separator: Separator string between before punctuation (keep_clause_breakers=True)
|
||||
keep_language_flags: True if language switching flags should be kept
|
||||
no_stress: True if stress characters should be removed
|
||||
|
||||
Returns:
|
||||
ipa - string of IPA phonemes
|
||||
"""
|
||||
self._maybe_init()
|
||||
|
||||
voice = voice or self.default_voice
|
||||
|
|
@ -97,13 +126,20 @@ class Phonemizer:
|
|||
Phonemizer.LANG_SWITCH_FLAG.sub("", line) for line in phoneme_lines
|
||||
]
|
||||
|
||||
if word_separator != " ":
|
||||
# Split/re-join words
|
||||
for line_idx in range(len(phoneme_lines)):
|
||||
phoneme_lines[line_idx] = word_separator.join(
|
||||
phoneme_lines[line_idx].split()
|
||||
)
|
||||
|
||||
# Re-insert clause breakers
|
||||
if missing_breakers:
|
||||
# pylint: disable=consider-using-enumerate
|
||||
for line_idx in range(len(phoneme_lines)):
|
||||
if line_idx < len(missing_breakers):
|
||||
phoneme_lines[line_idx] += (
|
||||
word_separator + missing_breakers[line_idx]
|
||||
punctuation_separator + missing_breakers[line_idx]
|
||||
)
|
||||
|
||||
phonemes_str = word_separator.join(line.strip() for line in phoneme_lines)
|
||||
|
|
|
|||
|
|
@ -1,14 +1,20 @@
|
|||
"""Command-line interface to espeak_phonemizer"""
|
||||
import argparse
|
||||
import csv
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
|
||||
from . import Phonemizer
|
||||
|
||||
_LOGGER = logging.getLogger("espeak_phonemizer")
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(prog="espeak_phonemizer")
|
||||
parser.add_argument("-v", "--voice", required=True, help="eSpeak voice to use")
|
||||
parser.add_argument("-v", "--voice", help="eSpeak voice to use")
|
||||
parser.add_argument(
|
||||
"-p", "--phoneme-separator", help="Separator character between phonemes"
|
||||
)
|
||||
|
|
@ -48,8 +54,30 @@ def main():
|
|||
parser.add_argument(
|
||||
"--csv-delimiter", default="|", help="Delimiter in CSV input and output"
|
||||
)
|
||||
parser.add_argument("--version", action="store_true", help="Print version and exit")
|
||||
parser.add_argument(
|
||||
"--debug", action="store_true", help="Print DEBUG messages to the console"
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.debug:
|
||||
logging.basicConfig(level=logging.DEBUG)
|
||||
else:
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
if args.version:
|
||||
# Print version and exit
|
||||
from . import __version__
|
||||
|
||||
print(__version__)
|
||||
sys.exit(0)
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
assert args.voice, "Missing -v/--voice"
|
||||
|
||||
if args.word_separator:
|
||||
assert (
|
||||
args.phoneme_separator.strip()
|
||||
|
|
@ -82,7 +110,7 @@ def main():
|
|||
phoneme_separator=args.phoneme_separator,
|
||||
keep_language_flags=args.keep_language_flags,
|
||||
no_stress=args.no_stress,
|
||||
punctuation_separator=args.phoneme_separator,
|
||||
punctuation_separator=args.phoneme_separator or "",
|
||||
)
|
||||
|
||||
if args.word_separator:
|
||||
|
|
|
|||
|
|
@ -1,11 +1,7 @@
|
|||
black==19.10b0
|
||||
coverage==5.0.4
|
||||
flake8==3.7.9
|
||||
mkdocs>=1.1
|
||||
mkdocs-material==5.1.1
|
||||
mypy==0.770
|
||||
pyinstaller==3.6
|
||||
pylint==2.4.4
|
||||
pytest==5.4.1
|
||||
pytest-cov==2.8.1
|
||||
yamllint==1.21.0
|
||||
|
|
|
|||
0
tests/__init__.py
Normal file
0
tests/__init__.py
Normal file
53
tests/test_phonemizer.py
Normal file
53
tests/test_phonemizer.py
Normal file
|
|
@ -0,0 +1,53 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Tests for Phonemizer class"""
|
||||
import unittest
|
||||
|
||||
from espeak_phonemizer import Phonemizer
|
||||
|
||||
|
||||
class PhonemizerTestCase(unittest.TestCase):
|
||||
"""Test cases for Phonemizer"""
|
||||
|
||||
def test_en(self):
|
||||
"""Test basic English"""
|
||||
phonemizer = Phonemizer(default_voice="en-us")
|
||||
phonemes = phonemizer.phonemize("test")
|
||||
self.assertEqual(phonemes, "tˈɛst")
|
||||
|
||||
def test_no_stress(self):
|
||||
"""Test stress removal"""
|
||||
phonemizer = Phonemizer(default_voice="en-us")
|
||||
phonemes = phonemizer.phonemize("test", no_stress=True)
|
||||
self.assertEqual(phonemes, "tɛst")
|
||||
|
||||
def test_phoneme_separator(self):
|
||||
"""Test with phoneme separator"""
|
||||
phonemizer = Phonemizer(default_voice="en-us")
|
||||
phonemes = phonemizer.phonemize("test", phoneme_separator="_")
|
||||
self.assertEqual(phonemes, "t_ˈɛ_s_t")
|
||||
|
||||
def test_word_phoneme_separators(self):
|
||||
"""Test with word and phoneme separators"""
|
||||
phonemizer = Phonemizer(default_voice="en-us")
|
||||
phonemes = phonemizer.phonemize(
|
||||
"test 1", phoneme_separator="_", word_separator="#"
|
||||
)
|
||||
self.assertEqual(phonemes, "t_ˈɛ_s_t#w_ˈʌ_n")
|
||||
|
||||
def test_keep_clause_breakers(self):
|
||||
"""Test keeping punctuation characters that break apart clauses"""
|
||||
phonemizer = Phonemizer(default_voice="en-us")
|
||||
phonemes = phonemizer.phonemize("test: 1, 2, 3!", keep_clause_breakers=True)
|
||||
self.assertEqual(phonemes, "tˈɛst: wˈʌn, tˈuː, θɹˈiː!")
|
||||
|
||||
def test_keep_language_flags(self):
|
||||
"""Test keeping language-switching flags"""
|
||||
phonemizer = Phonemizer(default_voice="fr")
|
||||
|
||||
# Without language flags
|
||||
phonemes = phonemizer.phonemize("library")
|
||||
self.assertEqual(phonemes, "lˈaɪbɹəɹi")
|
||||
|
||||
# With language flags
|
||||
phonemes = phonemizer.phonemize("library", keep_language_flags=True)
|
||||
self.assertEqual(phonemes, "(en)lˈaɪbɹəɹi(fr)")
|
||||
Loading…
Add table
Add a link
Reference in a new issue