Initial commit

This commit is contained in:
Michael Hansen 2021-08-02 17:19:20 -04:00
commit 38f1f4e126
18 changed files with 746 additions and 0 deletions

View file

@ -0,0 +1 @@
1.0

View file

@ -0,0 +1,139 @@
import ctypes
import re
import typing
class Phonemizer:
"""Use ctypes and libespeak-ng to get IPA phonemes from text"""
SEEK_SET = 0
EE_OK = 0
AUDIO_OUTPUT_SYNCHRONOUS = 0x02
espeakPHONEMES_IPA = 0x02
espeakCHARS_AUTO = 0
espeakPHONEMES = 0x100
LANG_SWITCH_FLAG = re.compile(r"\([^)]+\)")
DEFAULT_CLAUSE_BREAKERS = {",", ";", ":", ".", "!", "?"}
STRESS_PATTERN = re.compile(r"[ˈˌ]")
def __init__(
self,
default_voice: typing.Optional[str] = None,
clause_breakers: typing.Optional[typing.Collection[str]] = None,
):
self.current_voice: typing.Optional[str] = None
self.default_voice = default_voice
self.clause_breakers = clause_breakers or Phonemizer.DEFAULT_CLAUSE_BREAKERS
self.libc: typing.Any = None
self.lib_espeak: typing.Any = None
def phonemize(
self,
text: str,
voice: typing.Optional[str] = None,
keep_clause_breakers: bool = False,
phoneme_separator: typing.Optional[str] = None,
word_separator: str = " ",
punctuation_separator: str = "",
keep_language_flags: bool = False,
no_stress: bool = False,
) -> str:
"""Return IPA string for text"""
self._maybe_init()
voice = voice or self.default_voice
if (voice is not None) and (voice != self.current_voice):
self.current_voice = voice
voice_bytes = voice.encode("utf-8")
result = self.lib_espeak.espeak_SetVoiceByName(voice_bytes)
assert result == Phonemizer.EE_OK, f"Failed to set voice to {voice}"
missing_breakers = []
if keep_clause_breakers and self.clause_breakers:
missing_breakers = [c for c in text if c in self.clause_breakers]
# Create in-memory file for phoneme trace.
# espeak_TextToPhonemes segfaults no matter what I do, so this is the back.
phonemes_buffer = ctypes.c_char_p()
phonemes_size = ctypes.c_size_t()
phonemes_file = self.libc.open_memstream(
ctypes.byref(phonemes_buffer), ctypes.byref(phonemes_size)
)
try:
phoneme_flags = Phonemizer.espeakPHONEMES_IPA
if phoneme_separator:
phoneme_flags = phoneme_flags | (ord(phoneme_separator) << 8)
self.lib_espeak.espeak_SetPhonemeTrace(phoneme_flags, phonemes_file)
identifier = ctypes.c_uint()
user_data = ctypes.c_void_p()
text_bytes = text.encode("utf-8")
self.lib_espeak.espeak_Synth(
text_bytes,
0, # buflength
0, # position
0, # position_type
0, # end_position
Phonemizer.espeakCHARS_AUTO | Phonemizer.espeakPHONEMES,
identifier,
user_data,
)
self.libc.fflush(phonemes_file)
phoneme_lines = ctypes.string_at(phonemes_buffer).decode().splitlines()
if not keep_language_flags:
# Remove language switching flags, e.g. (en)
phoneme_lines = [
Phonemizer.LANG_SWITCH_FLAG.sub("", line) for line in phoneme_lines
]
# Re-insert clause breakers
if missing_breakers:
# pylint: disable=consider-using-enumerate
for line_idx in range(len(phoneme_lines)):
if line_idx < len(missing_breakers):
phoneme_lines[line_idx] += (
word_separator + missing_breakers[line_idx]
)
phonemes_str = word_separator.join(line.strip() for line in phoneme_lines)
if no_stress:
# Remove primary/secondary stress markers
phonemes_str = Phonemizer.STRESS_PATTERN.sub("", phonemes_str)
# Clean up multiple phoneme separators
if phoneme_separator:
phonemes_str = re.sub(
"[" + re.escape(phoneme_separator) + "]+",
phoneme_separator,
phonemes_str,
)
return phonemes_str
finally:
self.libc.fclose(phonemes_file)
def _maybe_init(self):
if self.libc and self.lib_espeak:
# Already initialize
return
self.libc = ctypes.cdll.LoadLibrary("libc.so.6")
self.libc.open_memstream.restype = ctypes.POINTER(ctypes.c_char)
self.lib_espeak = ctypes.cdll.LoadLibrary("libespeak-ng.so")
sample_rate = self.lib_espeak.espeak_Initialize(
Phonemizer.AUDIO_OUTPUT_SYNCHRONOUS, 0, None, 0
)
assert sample_rate > 0, "Failed to initialize libespeak-ng"

View file

@ -0,0 +1,100 @@
import argparse
import csv
import sys
from . import Phonemizer
def main():
parser = argparse.ArgumentParser(prog="espeak_phonemizer")
parser.add_argument("-v", "--voice", required=True, help="eSpeak voice to use")
parser.add_argument(
"-p", "--phoneme-separator", help="Separator character between phonemes"
)
parser.add_argument(
"-w",
"--word-separator",
help="Separator string between words (cannot be used if --phoneme-separator is whitespace)",
)
parser.add_argument(
"--keep-punctuation",
action="store_true",
help="Keep clause-breaking punctuation characters (,;:.!?)",
)
parser.add_argument(
"--keep-language-flags",
action="store_true",
help="Keep language-switching flags",
)
parser.add_argument(
"--print-input", action="store_true", help="Print input text before phonemes"
)
parser.add_argument(
"--output-separator",
default=" ",
help="Separator string between input text and phonemes",
)
parser.add_argument(
"--no-stress",
action="store_true",
help="Remove primary/secondary stress markers",
)
parser.add_argument(
"--csv",
action="store_true",
help="Input and output is CSV. Phonemes are added as a final column",
)
parser.add_argument(
"--csv-delimiter", default="|", help="Delimiter in CSV input and output"
)
args = parser.parse_args()
if args.word_separator:
assert (
args.phoneme_separator.strip()
), "Word separator cannot be used if phoneme separator is whitespace"
phonemizer = Phonemizer(default_voice=args.voice)
# CSV input/output
if args.csv:
csv_writer = csv.writer(sys.stdout, delimiter=args.csv_delimiter)
reader = csv.reader(sys.stdin, delimiter=args.csv_delimiter)
else:
csv_writer = None
reader = sys.stdin
for line in reader:
if args.csv:
text = line[-1]
else:
text = line.strip()
if not text:
continue
text_phonemes = phonemizer.phonemize(
text,
keep_clause_breakers=args.keep_punctuation,
phoneme_separator=args.phoneme_separator,
keep_language_flags=args.keep_language_flags,
no_stress=args.no_stress,
punctuation_separator=args.phoneme_separator,
)
if args.word_separator:
text_phonemes = args.word_separator.join(text_phonemes.split())
if args.csv:
assert csv_writer is not None
csv_writer.writerow((*line, text_phonemes))
else:
if args.print_input:
print(text, text_phonemes, sep=args.output_separator)
else:
print(text_phonemes)
# -----------------------------------------------------------------------------
if __name__ == "__main__":
main()

View file

@ -0,0 +1,244 @@
import argparse
import csv
import logging
import os
import sys
_LOGGER = logging.getLogger("phoneme_ids")
_STRESS = {"ˈ", "ˌ"}
_PUNCTUATION_MAP = {";": ",", ":": ",", "?": ".", "!": "."}
# -----------------------------------------------------------------------------
def main():
parser = argparse.ArgumentParser(prog="phoneme_ids")
parser.add_argument(
"--write-phonemes", help="Path to write phoneme ids text file (ID PHONEME)"
)
parser.add_argument(
"--read-phonemes", help="Read phoneme ids from a text file (ID PHONEME)"
)
parser.add_argument(
"-p", "--phoneme-separator", help="Separator character between phonemes"
)
parser.add_argument(
"-w", "--word-separator", default=" ", help="Separator character between words"
)
parser.add_argument(
"--id-separator", default=" ", help="Separator string each phoneme id"
)
parser.add_argument("--pad", help="Phoneme for padding (phoneme 0)")
parser.add_argument("--bos", help="Phoneme to put at beginning of sentence")
parser.add_argument("--eos", help="Phoneme to put at end of sentence")
parser.add_argument(
"--add-blank", action="store_true", help="Word separator is a phoneme"
)
parser.add_argument(
"--simple-punctuation",
action="store_true",
help="Map all punctuation into ',' and '.'",
)
parser.add_argument(
"--csv",
action="store_true",
help="Input and output is CSV. Phonemes ids are added as a final column",
)
parser.add_argument(
"--csv-delimiter", default="|", help="Delimiter in CSV input and output"
)
parser.add_argument(
"--output-separator",
default="|",
help="Separator string between input phonemes and phoneme ids",
)
parser.add_argument(
"--print-input", action="store_true", help="Print input text before phoneme ids"
)
parser.add_argument(
"--separate-stress",
action="store_true",
help="Pull primary/secondary stress out as separate phonemes",
)
args = parser.parse_args()
logging.basicConfig(level=logging.INFO)
phoneme_to_id = {}
if args.read_phonemes:
# Load from phonemes file
# Format is ID<space>PHONEME
with open(args.read_phonemes, "r") as phonemes_file:
for line in phonemes_file:
line = line.strip("\r\n")
if (not line) or line.startswith("#") or (" " not in line):
continue
parts = line.split(" ", maxsplit=1)
phoneme_id, phoneme = int(parts[0]), parts[1]
phoneme_to_id[phoneme] = phoneme_id
if args.pad and (args.pad not in phoneme_to_id):
# Add pad symbol
phoneme_to_id[args.pad] = len(phoneme_to_id)
if args.bos and (args.bos not in phoneme_to_id):
# Add BOS symbol
phoneme_to_id[args.bos] = len(phoneme_to_id)
if args.eos and (args.eos not in phoneme_to_id):
# Add EOS symbol
phoneme_to_id[args.eos] = len(phoneme_to_id)
if args.add_blank:
# Word separator itself is a phoneme
if args.word_separator not in phoneme_to_id:
phoneme_to_id[args.word_separator] = len(phoneme_to_id)
word_sep_id = phoneme_to_id[args.word_separator]
word_sep_str = f" {word_sep_id} "
else:
word_sep_str = args.word_separator
if args.separate_stress:
# Add stress symbols
for stress in sorted(_STRESS):
if stress not in phoneme_to_id:
phoneme_to_id[stress] = len(phoneme_to_id)
# -------------------------------------------------------------------------
if os.isatty(sys.stdin.fileno()):
print("Reading from stdin...", file=sys.stderr)
# CSV input/output
if args.csv:
csv_writer = csv.writer(sys.stdout, delimiter=args.csv_delimiter)
reader = csv.reader(sys.stdin, delimiter=args.csv_delimiter)
else:
csv_writer = None
reader = sys.stdin
# Read all input and get set of phonemes
all_phonemes = set(phoneme_to_id.keys())
if args.simple_punctuation:
# Add , and .
all_phonemes.update(sorted(_PUNCTUATION_MAP.values()))
lines = []
for line in reader:
if args.csv:
phonemes_str = line[-1]
else:
phonemes_str = line.strip()
if not phonemes_str:
continue
# Split into words
if args.phoneme_separator:
word_phonemes = [
word.split(args.phoneme_separator)
for word in phonemes_str.split(args.word_separator)
]
else:
word_phonemes = [
list(word) for word in phonemes_str.split(args.word_separator)
]
lines.append((line, word_phonemes))
for word in word_phonemes:
for phoneme in word:
if args.separate_stress:
# Split stress out
while phoneme and (phoneme[0] in _STRESS):
phoneme = phoneme[1:]
if phoneme:
if args.simple_punctuation:
phoneme = _PUNCTUATION_MAP.get(phoneme, phoneme)
all_phonemes.add(phoneme)
# Assign phonemes to ids in sorted order
for phoneme in sorted(all_phonemes):
if phoneme not in phoneme_to_id:
phoneme_to_id[phoneme] = len(phoneme_to_id)
# -------------------------------------------------------------------------
for line, word_phonemes in lines:
if args.csv:
phonemes_str = line[-1]
else:
phonemes_str = line.strip()
# Transform into phoneme ids
word_phoneme_ids = []
# Add beginning-of-sentence symbol
if args.bos:
word_phoneme_ids.append([phoneme_to_id[args.bos]])
for word in word_phonemes:
word_ids = []
for phoneme in word:
if args.separate_stress:
# Split stress out
while phoneme and (phoneme[0] in _STRESS):
stress = phoneme[0]
word_ids.append(phoneme_to_id[stress])
phoneme = phoneme[1:]
if phoneme:
if args.simple_punctuation:
phoneme = _PUNCTUATION_MAP.get(phoneme, phoneme)
word_ids.append(phoneme_to_id[phoneme])
if word_ids:
word_phoneme_ids.append(word_ids)
# Add end-of-sentence symbol
if args.eos:
word_phoneme_ids.append([phoneme_to_id[args.eos]])
phoneme_ids_str = word_sep_str.join(
(
args.id_separator.join((str(p_id) for p_id in word))
for word in word_phoneme_ids
)
)
if args.csv:
# Add phoneme ids as last column
assert csv_writer is not None
csv_writer.writerow((*line, phoneme_ids_str))
else:
if args.print_input:
# Print input phonemes as well as phoneme ids
print(phonemes_str, phoneme_ids_str, sep=args.output_separator)
else:
# Just print phoneme ids
print(phoneme_ids_str)
# -------------------------------------------------------------------------
if args.write_phonemes:
# Write file with ID<space>PHONEME format
with open(args.write_phonemes, "w") as phonemes_file:
for phoneme, phoneme_id in sorted(
phoneme_to_id.items(), key=lambda kv: kv[1]
):
print(phoneme_id, phoneme, file=phonemes_file)
# -----------------------------------------------------------------------------
if __name__ == "__main__":
main()

View file