Initial commit

This commit is contained in:
Michael Hansen 2021-08-02 17:19:20 -04:00
commit 38f1f4e126
18 changed files with 746 additions and 0 deletions

View file

@ -0,0 +1,139 @@
import ctypes
import re
import typing
class Phonemizer:
"""Use ctypes and libespeak-ng to get IPA phonemes from text"""
SEEK_SET = 0
EE_OK = 0
AUDIO_OUTPUT_SYNCHRONOUS = 0x02
espeakPHONEMES_IPA = 0x02
espeakCHARS_AUTO = 0
espeakPHONEMES = 0x100
LANG_SWITCH_FLAG = re.compile(r"\([^)]+\)")
DEFAULT_CLAUSE_BREAKERS = {",", ";", ":", ".", "!", "?"}
STRESS_PATTERN = re.compile(r"[ˈˌ]")
def __init__(
self,
default_voice: typing.Optional[str] = None,
clause_breakers: typing.Optional[typing.Collection[str]] = None,
):
self.current_voice: typing.Optional[str] = None
self.default_voice = default_voice
self.clause_breakers = clause_breakers or Phonemizer.DEFAULT_CLAUSE_BREAKERS
self.libc: typing.Any = None
self.lib_espeak: typing.Any = None
def phonemize(
self,
text: str,
voice: typing.Optional[str] = None,
keep_clause_breakers: bool = False,
phoneme_separator: typing.Optional[str] = None,
word_separator: str = " ",
punctuation_separator: str = "",
keep_language_flags: bool = False,
no_stress: bool = False,
) -> str:
"""Return IPA string for text"""
self._maybe_init()
voice = voice or self.default_voice
if (voice is not None) and (voice != self.current_voice):
self.current_voice = voice
voice_bytes = voice.encode("utf-8")
result = self.lib_espeak.espeak_SetVoiceByName(voice_bytes)
assert result == Phonemizer.EE_OK, f"Failed to set voice to {voice}"
missing_breakers = []
if keep_clause_breakers and self.clause_breakers:
missing_breakers = [c for c in text if c in self.clause_breakers]
# Create in-memory file for phoneme trace.
# espeak_TextToPhonemes segfaults no matter what I do, so this is the back.
phonemes_buffer = ctypes.c_char_p()
phonemes_size = ctypes.c_size_t()
phonemes_file = self.libc.open_memstream(
ctypes.byref(phonemes_buffer), ctypes.byref(phonemes_size)
)
try:
phoneme_flags = Phonemizer.espeakPHONEMES_IPA
if phoneme_separator:
phoneme_flags = phoneme_flags | (ord(phoneme_separator) << 8)
self.lib_espeak.espeak_SetPhonemeTrace(phoneme_flags, phonemes_file)
identifier = ctypes.c_uint()
user_data = ctypes.c_void_p()
text_bytes = text.encode("utf-8")
self.lib_espeak.espeak_Synth(
text_bytes,
0, # buflength
0, # position
0, # position_type
0, # end_position
Phonemizer.espeakCHARS_AUTO | Phonemizer.espeakPHONEMES,
identifier,
user_data,
)
self.libc.fflush(phonemes_file)
phoneme_lines = ctypes.string_at(phonemes_buffer).decode().splitlines()
if not keep_language_flags:
# Remove language switching flags, e.g. (en)
phoneme_lines = [
Phonemizer.LANG_SWITCH_FLAG.sub("", line) for line in phoneme_lines
]
# Re-insert clause breakers
if missing_breakers:
# pylint: disable=consider-using-enumerate
for line_idx in range(len(phoneme_lines)):
if line_idx < len(missing_breakers):
phoneme_lines[line_idx] += (
word_separator + missing_breakers[line_idx]
)
phonemes_str = word_separator.join(line.strip() for line in phoneme_lines)
if no_stress:
# Remove primary/secondary stress markers
phonemes_str = Phonemizer.STRESS_PATTERN.sub("", phonemes_str)
# Clean up multiple phoneme separators
if phoneme_separator:
phonemes_str = re.sub(
"[" + re.escape(phoneme_separator) + "]+",
phoneme_separator,
phonemes_str,
)
return phonemes_str
finally:
self.libc.fclose(phonemes_file)
def _maybe_init(self):
if self.libc and self.lib_espeak:
# Already initialize
return
self.libc = ctypes.cdll.LoadLibrary("libc.so.6")
self.libc.open_memstream.restype = ctypes.POINTER(ctypes.c_char)
self.lib_espeak = ctypes.cdll.LoadLibrary("libespeak-ng.so")
sample_rate = self.lib_espeak.espeak_Initialize(
Phonemizer.AUDIO_OUTPUT_SYNCHRONOUS, 0, None, 0
)
assert sample_rate > 0, "Failed to initialize libespeak-ng"