commit 38f1f4e1264c45fc777fcdd134ccc21b1713948a Author: Michael Hansen Date: Mon Aug 2 17:19:20 2021 -0400 Initial commit diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..83214be --- /dev/null +++ b/.gitignore @@ -0,0 +1,10 @@ +__pycache__/ +.mypy_cache/ +*.egg-info/ +/build/ +/.venv/ +/dist/ +/download/ +/gruut/ +*.onnx +.dockerignore \ No newline at end of file diff --git a/.isort.cfg b/.isort.cfg new file mode 100644 index 0000000..ba2778d --- /dev/null +++ b/.isort.cfg @@ -0,0 +1,6 @@ +[settings] +multi_line_output=3 +include_trailing_comma=True +force_grid_wrap=0 +use_parentheses=True +line_length=88 diff --git a/.projectile b/.projectile new file mode 100644 index 0000000..1cdab6b --- /dev/null +++ b/.projectile @@ -0,0 +1,2 @@ +- /.venv/ +- /.mypy_cache/ \ No newline at end of file diff --git a/Makefile b/Makefile new file mode 100644 index 0000000..b896fa5 --- /dev/null +++ b/Makefile @@ -0,0 +1,29 @@ +SHELL := bash + +.PHONY: check clean reformat dist docker amd64 index + +all: dist + +venv: + scripts/create-venv.sh + +check: + scripts/check-code.sh + +reformat: + scripts/format-code.sh + +dist: + python3 setup.py sdist + +docker: + scripts/build-docker.sh + for lang in de-de en-us es-es fr-fr it-it nl ru-ru sv-se; do \ + LARYNX_LANGUAGE=$$lang scripts/build-docker.sh; \ + done + +amd64: + NOBUILDX=1 scripts/build-docker.sh + +index: + bin/make_sample_html.py local/ > index.html diff --git a/README.md b/README.md new file mode 100644 index 0000000..304d4c1 --- /dev/null +++ b/README.md @@ -0,0 +1 @@ +# eSpeak Phonemizer diff --git a/espeak_phonemizer/VERSION b/espeak_phonemizer/VERSION new file mode 100644 index 0000000..9f8e9b6 --- /dev/null +++ b/espeak_phonemizer/VERSION @@ -0,0 +1 @@ +1.0 \ No newline at end of file diff --git a/espeak_phonemizer/__init__.py b/espeak_phonemizer/__init__.py new file mode 100644 index 0000000..de6e071 --- /dev/null +++ b/espeak_phonemizer/__init__.py @@ -0,0 +1,139 @@ +import ctypes +import re +import typing + + +class Phonemizer: + """Use ctypes and libespeak-ng to get IPA phonemes from text""" + + SEEK_SET = 0 + + EE_OK = 0 + + AUDIO_OUTPUT_SYNCHRONOUS = 0x02 + espeakPHONEMES_IPA = 0x02 + espeakCHARS_AUTO = 0 + espeakPHONEMES = 0x100 + + LANG_SWITCH_FLAG = re.compile(r"\([^)]+\)") + + DEFAULT_CLAUSE_BREAKERS = {",", ";", ":", ".", "!", "?"} + + STRESS_PATTERN = re.compile(r"[ˈˌ]") + + def __init__( + self, + default_voice: typing.Optional[str] = None, + clause_breakers: typing.Optional[typing.Collection[str]] = None, + ): + self.current_voice: typing.Optional[str] = None + self.default_voice = default_voice + self.clause_breakers = clause_breakers or Phonemizer.DEFAULT_CLAUSE_BREAKERS + + self.libc: typing.Any = None + self.lib_espeak: typing.Any = None + + def phonemize( + self, + text: str, + voice: typing.Optional[str] = None, + keep_clause_breakers: bool = False, + phoneme_separator: typing.Optional[str] = None, + word_separator: str = " ", + punctuation_separator: str = "", + keep_language_flags: bool = False, + no_stress: bool = False, + ) -> str: + """Return IPA string for text""" + self._maybe_init() + + voice = voice or self.default_voice + + if (voice is not None) and (voice != self.current_voice): + self.current_voice = voice + voice_bytes = voice.encode("utf-8") + result = self.lib_espeak.espeak_SetVoiceByName(voice_bytes) + assert result == Phonemizer.EE_OK, f"Failed to set voice to {voice}" + + missing_breakers = [] + if keep_clause_breakers and self.clause_breakers: + missing_breakers = [c for c in text if c in self.clause_breakers] + + # Create in-memory file for phoneme trace. + # espeak_TextToPhonemes segfaults no matter what I do, so this is the back. + phonemes_buffer = ctypes.c_char_p() + phonemes_size = ctypes.c_size_t() + phonemes_file = self.libc.open_memstream( + ctypes.byref(phonemes_buffer), ctypes.byref(phonemes_size) + ) + + try: + phoneme_flags = Phonemizer.espeakPHONEMES_IPA + if phoneme_separator: + phoneme_flags = phoneme_flags | (ord(phoneme_separator) << 8) + + self.lib_espeak.espeak_SetPhonemeTrace(phoneme_flags, phonemes_file) + + identifier = ctypes.c_uint() + user_data = ctypes.c_void_p() + text_bytes = text.encode("utf-8") + self.lib_espeak.espeak_Synth( + text_bytes, + 0, # buflength + 0, # position + 0, # position_type + 0, # end_position + Phonemizer.espeakCHARS_AUTO | Phonemizer.espeakPHONEMES, + identifier, + user_data, + ) + self.libc.fflush(phonemes_file) + + phoneme_lines = ctypes.string_at(phonemes_buffer).decode().splitlines() + + if not keep_language_flags: + # Remove language switching flags, e.g. (en) + phoneme_lines = [ + Phonemizer.LANG_SWITCH_FLAG.sub("", line) for line in phoneme_lines + ] + + # Re-insert clause breakers + if missing_breakers: + # pylint: disable=consider-using-enumerate + for line_idx in range(len(phoneme_lines)): + if line_idx < len(missing_breakers): + phoneme_lines[line_idx] += ( + word_separator + missing_breakers[line_idx] + ) + + phonemes_str = word_separator.join(line.strip() for line in phoneme_lines) + + if no_stress: + # Remove primary/secondary stress markers + phonemes_str = Phonemizer.STRESS_PATTERN.sub("", phonemes_str) + + # Clean up multiple phoneme separators + if phoneme_separator: + phonemes_str = re.sub( + "[" + re.escape(phoneme_separator) + "]+", + phoneme_separator, + phonemes_str, + ) + + return phonemes_str + finally: + self.libc.fclose(phonemes_file) + + def _maybe_init(self): + if self.libc and self.lib_espeak: + # Already initialize + return + + self.libc = ctypes.cdll.LoadLibrary("libc.so.6") + self.libc.open_memstream.restype = ctypes.POINTER(ctypes.c_char) + + self.lib_espeak = ctypes.cdll.LoadLibrary("libespeak-ng.so") + sample_rate = self.lib_espeak.espeak_Initialize( + Phonemizer.AUDIO_OUTPUT_SYNCHRONOUS, 0, None, 0 + ) + assert sample_rate > 0, "Failed to initialize libespeak-ng" diff --git a/espeak_phonemizer/__main__.py b/espeak_phonemizer/__main__.py new file mode 100644 index 0000000..b4b37df --- /dev/null +++ b/espeak_phonemizer/__main__.py @@ -0,0 +1,100 @@ +import argparse +import csv +import sys + +from . import Phonemizer + + +def main(): + parser = argparse.ArgumentParser(prog="espeak_phonemizer") + parser.add_argument("-v", "--voice", required=True, help="eSpeak voice to use") + parser.add_argument( + "-p", "--phoneme-separator", help="Separator character between phonemes" + ) + parser.add_argument( + "-w", + "--word-separator", + help="Separator string between words (cannot be used if --phoneme-separator is whitespace)", + ) + parser.add_argument( + "--keep-punctuation", + action="store_true", + help="Keep clause-breaking punctuation characters (,;:.!?)", + ) + parser.add_argument( + "--keep-language-flags", + action="store_true", + help="Keep language-switching flags", + ) + parser.add_argument( + "--print-input", action="store_true", help="Print input text before phonemes" + ) + parser.add_argument( + "--output-separator", + default=" ", + help="Separator string between input text and phonemes", + ) + parser.add_argument( + "--no-stress", + action="store_true", + help="Remove primary/secondary stress markers", + ) + parser.add_argument( + "--csv", + action="store_true", + help="Input and output is CSV. Phonemes are added as a final column", + ) + parser.add_argument( + "--csv-delimiter", default="|", help="Delimiter in CSV input and output" + ) + args = parser.parse_args() + + if args.word_separator: + assert ( + args.phoneme_separator.strip() + ), "Word separator cannot be used if phoneme separator is whitespace" + + phonemizer = Phonemizer(default_voice=args.voice) + + # CSV input/output + if args.csv: + csv_writer = csv.writer(sys.stdout, delimiter=args.csv_delimiter) + reader = csv.reader(sys.stdin, delimiter=args.csv_delimiter) + else: + csv_writer = None + reader = sys.stdin + + for line in reader: + if args.csv: + text = line[-1] + else: + text = line.strip() + if not text: + continue + + text_phonemes = phonemizer.phonemize( + text, + keep_clause_breakers=args.keep_punctuation, + phoneme_separator=args.phoneme_separator, + keep_language_flags=args.keep_language_flags, + no_stress=args.no_stress, + punctuation_separator=args.phoneme_separator, + ) + + if args.word_separator: + text_phonemes = args.word_separator.join(text_phonemes.split()) + + if args.csv: + assert csv_writer is not None + csv_writer.writerow((*line, text_phonemes)) + else: + if args.print_input: + print(text, text_phonemes, sep=args.output_separator) + else: + print(text_phonemes) + + +# ----------------------------------------------------------------------------- + +if __name__ == "__main__": + main() diff --git a/espeak_phonemizer/phoneme_ids.py b/espeak_phonemizer/phoneme_ids.py new file mode 100644 index 0000000..477ec9f --- /dev/null +++ b/espeak_phonemizer/phoneme_ids.py @@ -0,0 +1,244 @@ +import argparse +import csv +import logging +import os +import sys + +_LOGGER = logging.getLogger("phoneme_ids") + +_STRESS = {"ˈ", "ˌ"} + +_PUNCTUATION_MAP = {";": ",", ":": ",", "?": ".", "!": "."} + +# ----------------------------------------------------------------------------- + + +def main(): + parser = argparse.ArgumentParser(prog="phoneme_ids") + parser.add_argument( + "--write-phonemes", help="Path to write phoneme ids text file (ID PHONEME)" + ) + parser.add_argument( + "--read-phonemes", help="Read phoneme ids from a text file (ID PHONEME)" + ) + parser.add_argument( + "-p", "--phoneme-separator", help="Separator character between phonemes" + ) + parser.add_argument( + "-w", "--word-separator", default=" ", help="Separator character between words" + ) + parser.add_argument( + "--id-separator", default=" ", help="Separator string each phoneme id" + ) + parser.add_argument("--pad", help="Phoneme for padding (phoneme 0)") + parser.add_argument("--bos", help="Phoneme to put at beginning of sentence") + parser.add_argument("--eos", help="Phoneme to put at end of sentence") + parser.add_argument( + "--add-blank", action="store_true", help="Word separator is a phoneme" + ) + parser.add_argument( + "--simple-punctuation", + action="store_true", + help="Map all punctuation into ',' and '.'", + ) + parser.add_argument( + "--csv", + action="store_true", + help="Input and output is CSV. Phonemes ids are added as a final column", + ) + parser.add_argument( + "--csv-delimiter", default="|", help="Delimiter in CSV input and output" + ) + parser.add_argument( + "--output-separator", + default="|", + help="Separator string between input phonemes and phoneme ids", + ) + parser.add_argument( + "--print-input", action="store_true", help="Print input text before phoneme ids" + ) + parser.add_argument( + "--separate-stress", + action="store_true", + help="Pull primary/secondary stress out as separate phonemes", + ) + args = parser.parse_args() + + logging.basicConfig(level=logging.INFO) + + phoneme_to_id = {} + + if args.read_phonemes: + # Load from phonemes file + # Format is IDPHONEME + with open(args.read_phonemes, "r") as phonemes_file: + for line in phonemes_file: + line = line.strip("\r\n") + if (not line) or line.startswith("#") or (" " not in line): + continue + + parts = line.split(" ", maxsplit=1) + phoneme_id, phoneme = int(parts[0]), parts[1] + phoneme_to_id[phoneme] = phoneme_id + + if args.pad and (args.pad not in phoneme_to_id): + # Add pad symbol + phoneme_to_id[args.pad] = len(phoneme_to_id) + + if args.bos and (args.bos not in phoneme_to_id): + # Add BOS symbol + phoneme_to_id[args.bos] = len(phoneme_to_id) + + if args.eos and (args.eos not in phoneme_to_id): + # Add EOS symbol + phoneme_to_id[args.eos] = len(phoneme_to_id) + + if args.add_blank: + # Word separator itself is a phoneme + if args.word_separator not in phoneme_to_id: + phoneme_to_id[args.word_separator] = len(phoneme_to_id) + + word_sep_id = phoneme_to_id[args.word_separator] + word_sep_str = f" {word_sep_id} " + else: + word_sep_str = args.word_separator + + if args.separate_stress: + # Add stress symbols + for stress in sorted(_STRESS): + if stress not in phoneme_to_id: + phoneme_to_id[stress] = len(phoneme_to_id) + + # ------------------------------------------------------------------------- + + if os.isatty(sys.stdin.fileno()): + print("Reading from stdin...", file=sys.stderr) + + # CSV input/output + if args.csv: + csv_writer = csv.writer(sys.stdout, delimiter=args.csv_delimiter) + reader = csv.reader(sys.stdin, delimiter=args.csv_delimiter) + else: + csv_writer = None + reader = sys.stdin + + # Read all input and get set of phonemes + all_phonemes = set(phoneme_to_id.keys()) + + if args.simple_punctuation: + # Add , and . + all_phonemes.update(sorted(_PUNCTUATION_MAP.values())) + + lines = [] + + for line in reader: + if args.csv: + phonemes_str = line[-1] + else: + phonemes_str = line.strip() + if not phonemes_str: + continue + + # Split into words + if args.phoneme_separator: + word_phonemes = [ + word.split(args.phoneme_separator) + for word in phonemes_str.split(args.word_separator) + ] + else: + word_phonemes = [ + list(word) for word in phonemes_str.split(args.word_separator) + ] + + lines.append((line, word_phonemes)) + + for word in word_phonemes: + for phoneme in word: + if args.separate_stress: + # Split stress out + while phoneme and (phoneme[0] in _STRESS): + phoneme = phoneme[1:] + + if phoneme: + if args.simple_punctuation: + phoneme = _PUNCTUATION_MAP.get(phoneme, phoneme) + + all_phonemes.add(phoneme) + + # Assign phonemes to ids in sorted order + for phoneme in sorted(all_phonemes): + if phoneme not in phoneme_to_id: + phoneme_to_id[phoneme] = len(phoneme_to_id) + + # ------------------------------------------------------------------------- + + for line, word_phonemes in lines: + if args.csv: + phonemes_str = line[-1] + else: + phonemes_str = line.strip() + + # Transform into phoneme ids + word_phoneme_ids = [] + + # Add beginning-of-sentence symbol + if args.bos: + word_phoneme_ids.append([phoneme_to_id[args.bos]]) + + for word in word_phonemes: + word_ids = [] + for phoneme in word: + if args.separate_stress: + # Split stress out + while phoneme and (phoneme[0] in _STRESS): + stress = phoneme[0] + word_ids.append(phoneme_to_id[stress]) + phoneme = phoneme[1:] + + if phoneme: + if args.simple_punctuation: + phoneme = _PUNCTUATION_MAP.get(phoneme, phoneme) + + word_ids.append(phoneme_to_id[phoneme]) + + if word_ids: + word_phoneme_ids.append(word_ids) + + # Add end-of-sentence symbol + if args.eos: + word_phoneme_ids.append([phoneme_to_id[args.eos]]) + + phoneme_ids_str = word_sep_str.join( + ( + args.id_separator.join((str(p_id) for p_id in word)) + for word in word_phoneme_ids + ) + ) + + if args.csv: + # Add phoneme ids as last column + assert csv_writer is not None + csv_writer.writerow((*line, phoneme_ids_str)) + else: + if args.print_input: + # Print input phonemes as well as phoneme ids + print(phonemes_str, phoneme_ids_str, sep=args.output_separator) + else: + # Just print phoneme ids + print(phoneme_ids_str) + + # ------------------------------------------------------------------------- + + if args.write_phonemes: + # Write file with IDPHONEME format + with open(args.write_phonemes, "w") as phonemes_file: + for phoneme, phoneme_id in sorted( + phoneme_to_id.items(), key=lambda kv: kv[1] + ): + print(phoneme_id, phoneme, file=phonemes_file) + + +# ----------------------------------------------------------------------------- + +if __name__ == "__main__": + main() diff --git a/espeak_phonemizer/py.typed b/espeak_phonemizer/py.typed new file mode 100644 index 0000000..e69de29 diff --git a/mypy.ini b/mypy.ini new file mode 100644 index 0000000..e2adf1b --- /dev/null +++ b/mypy.ini @@ -0,0 +1,3 @@ + +[mypy] + diff --git a/pylintrc b/pylintrc new file mode 100644 index 0000000..37b5825 --- /dev/null +++ b/pylintrc @@ -0,0 +1,40 @@ +[MESSAGES CONTROL] +disable= + format, + abstract-class-little-used, + abstract-method, + cyclic-import, + duplicate-code, + global-statement, + import-outside-toplevel, + inconsistent-return-statements, + locally-disabled, + not-context-manager, + redefined-variable-type, + too-few-public-methods, + too-many-arguments, + too-many-branches, + too-many-instance-attributes, + too-many-lines, + too-many-locals, + too-many-public-methods, + too-many-return-statements, + too-many-statements, + too-many-boolean-expressions, + unnecessary-pass, + unused-argument, + broad-except, + too-many-nested-blocks, + invalid-name, + no-self-use, + missing-function-docstring, + missing-module-docstring, + missing-class-docstring, + consider-using-enumerate, + fixme + +[FORMAT] +expected-line-ending-format=LF + +[TYPECHECK] +generated-members=torch.* \ No newline at end of file diff --git a/requirements_dev.txt b/requirements_dev.txt new file mode 100644 index 0000000..464fa84 --- /dev/null +++ b/requirements_dev.txt @@ -0,0 +1,11 @@ +black==19.10b0 +coverage==5.0.4 +flake8==3.7.9 +mkdocs>=1.1 +mkdocs-material==5.1.1 +mypy==0.770 +pyinstaller==3.6 +pylint==2.4.4 +pytest==5.4.1 +pytest-cov==2.8.1 +yamllint==1.21.0 diff --git a/scripts/check-code.sh b/scripts/check-code.sh new file mode 100755 index 0000000..6929a60 --- /dev/null +++ b/scripts/check-code.sh @@ -0,0 +1,27 @@ +#!/usr/bin/env bash +set -e + +# Directory of *this* script +this_dir="$( cd "$( dirname "$0" )" && pwd )" +src_dir="$(realpath "${this_dir}/..")" + +venv="${src_dir}/.venv" +if [[ -d "${venv}" ]]; then + source "${venv}/bin/activate" +fi + +python_files=("${src_dir}/espeak_phonemizer/"*.py) + +export PYTHONPATH="${src_dir}" + +# ----------------------------------------------------------------------------- + +flake8 "${python_files[@]}" +pylint "${python_files[@]}" +mypy "${python_files[@]}" +black --check "${python_files[@]}" +isort --check-only "${python_files[@]}" + +# ----------------------------------------------------------------------------- + +echo "OK" diff --git a/scripts/create-venv.sh b/scripts/create-venv.sh new file mode 100755 index 0000000..a821eb5 --- /dev/null +++ b/scripts/create-venv.sh @@ -0,0 +1,48 @@ +#!/usr/bin/env bash +set -e + +: "${PIP_INSTALL=install}" +: "${PIP_VERSION=pip}" + +# Directory of *this* script +this_dir="$( cd "$( dirname "$0" )" && pwd )" +src_dir="$(realpath "${this_dir}/..")" + +# ----------------------------------------------------------------------------- + +venv="${src_dir}/.venv" + +# ----------------------------------------------------------------------------- + +: "${PYTHON=python3}" + +python_version="$(${PYTHON} --version)" + +# Create virtual environment +echo "Creating virtual environment at ${venv} (${python_version})" +rm -rf "${venv}" +"${PYTHON}" -m venv "${venv}" +source "${venv}/bin/activate" + +# Install Python dependencies +echo 'Installing Python dependencies' +pip3 ${PIP_INSTALL} --upgrade "${PIP_VERSION}" +pip3 ${PIP_INSTALL} --upgrade wheel setuptools + +if [[ -n "${PIP_PREINSTALL_PACKAGES}" ]]; then + pip3 ${PIP_INSTALL} ${PIP_PREINSTALL_PACKAGES} +fi + +if [[ -f requirements.txt ]]; then + pip3 ${PIP_INSTALL} -r requirements.txt +fi + + +# Development dependencies +if [[ -f requirements_dev.txt ]]; then + pip3 ${PIP_INSTALL} -r requirements_dev.txt || echo "Failed to install development dependencies" >&2 +fi + +# ----------------------------------------------------------------------------- + +echo "OK" diff --git a/scripts/format-code.sh b/scripts/format-code.sh new file mode 100755 index 0000000..71be7ff --- /dev/null +++ b/scripts/format-code.sh @@ -0,0 +1,24 @@ +#!/usr/bin/env bash +set -e + +# Directory of *this* script +this_dir="$( cd "$( dirname "$0" )" && pwd )" +src_dir="$(realpath "${this_dir}/..")" + +venv="${src_dir}/.venv" +if [[ -d "${venv}" ]]; then + source "${venv}/bin/activate" +fi + +python_files=("${src_dir}/espeak_phonemizer/"*.py) + +export PYTHONPATH="${src_dir}" + +# ----------------------------------------------------------------------------- + +black "${python_files[@]}" +isort "${python_files[@]}" + +# ----------------------------------------------------------------------------- + +echo "OK" diff --git a/setup.cfg b/setup.cfg new file mode 100644 index 0000000..830e0a9 --- /dev/null +++ b/setup.cfg @@ -0,0 +1,14 @@ +[flake8] +# To work with Black +max-line-length = 88 +# E501: line too long +# W503: Line break occurred before a binary operator +# E203: Whitespace before ':' +# D202 No blank lines allowed after function docstring +# W504 line break after binary operator +ignore = + E501, + W503, + E203, + D202, + W504 diff --git a/setup.py b/setup.py new file mode 100644 index 0000000..5301b73 --- /dev/null +++ b/setup.py @@ -0,0 +1,47 @@ +"""Setup file for espeak_phonemizer""" +import os +from pathlib import Path + +import setuptools + +this_dir = Path(__file__).parent +module_dir = this_dir / "espeak_phonemizer" + +# ----------------------------------------------------------------------------- + +# Load README in as long description +long_description: str = "" +readme_path = this_dir / "README.md" +if readme_path.is_file(): + long_description = readme_path.read_text() + +version_path = module_dir / "VERSION" +with open(version_path, "r") as version_file: + version = version_file.read().strip() + +# ----------------------------------------------------------------------------- + +setuptools.setup( + name="espeak_phonemize", + version=version, + description="Lightweight International Phonetic Alphabet (IPA) phonemizer that uses libespeak-ng", + author="Michael Hansen", + author_email="mike@rhasspy.org", + url="https://github.com/synesthesiam/espeak-phonemizer", + packages=setuptools.find_packages(), + package_data={"espeak_phonemizer": ["VERSION", "py.typed"]}, + install_requires=requirements, + entry_points={ + "console_scripts": [ + "espeak-phonemizer = espeak_phonemizer.__main__:main", + ] + }, + classifiers=[ + "Programming Language :: Python :: 3", + "Programming Language :: Python :: 3.7", + "Programming Language :: Python :: 3.8", + "License :: OSI Approved :: MIT License", + ], + long_description=long_description, + long_description_content_type="text/markdown", +)