178 lines
6.2 KiB
Python
178 lines
6.2 KiB
Python
"""Uses ctypes and libespeak-ng to get IPA phonemes from text"""
|
|
import ctypes
|
|
import re
|
|
import typing
|
|
from pathlib import Path
|
|
|
|
_DIR = Path(__file__).parent
|
|
__version__ = (_DIR / "VERSION").read_text().strip()
|
|
|
|
# -----------------------------------------------------------------------------
|
|
|
|
|
|
class Phonemizer:
|
|
"""
|
|
Use ctypes and libespeak-ng to get IPA phonemes from text.
|
|
Not thread safe.
|
|
|
|
Requires libc.so.6
|
|
Tries to use libespeak-ng.so or libespeak-ng.so.1
|
|
"""
|
|
|
|
SEEK_SET = 0
|
|
|
|
EE_OK = 0
|
|
|
|
AUDIO_OUTPUT_SYNCHRONOUS = 0x02
|
|
espeakPHONEMES_IPA = 0x02
|
|
espeakCHARS_AUTO = 0
|
|
espeakPHONEMES = 0x100
|
|
|
|
LANG_SWITCH_FLAG = re.compile(r"\([^)]*\)")
|
|
|
|
DEFAULT_CLAUSE_BREAKERS = {",", ";", ":", ".", "!", "?"}
|
|
|
|
STRESS_PATTERN = re.compile(r"[ˈˌ]")
|
|
|
|
def __init__(
|
|
self,
|
|
default_voice: typing.Optional[str] = None,
|
|
clause_breakers: typing.Optional[typing.Collection[str]] = None,
|
|
):
|
|
self.current_voice: typing.Optional[str] = None
|
|
self.default_voice = default_voice
|
|
self.clause_breakers = clause_breakers or Phonemizer.DEFAULT_CLAUSE_BREAKERS
|
|
|
|
self.libc: typing.Any = None
|
|
self.lib_espeak: typing.Any = None
|
|
|
|
def phonemize(
|
|
self,
|
|
text: str,
|
|
voice: typing.Optional[str] = None,
|
|
keep_clause_breakers: bool = False,
|
|
phoneme_separator: typing.Optional[str] = None,
|
|
word_separator: str = " ",
|
|
punctuation_separator: str = "",
|
|
keep_language_flags: bool = False,
|
|
no_stress: bool = False,
|
|
) -> str:
|
|
"""
|
|
Return IPA string for text.
|
|
Not thread safe.
|
|
|
|
Args:
|
|
text: Text to phonemize
|
|
voice: optional voice (uses self.default_voice if None)
|
|
keep_clause_breakers: True if punctuation symbols should be kept
|
|
phoneme_separator: Separator character between phonemes
|
|
word_separator: Separator string between words (default: space)
|
|
punctuation_separator: Separator string between before punctuation (keep_clause_breakers=True)
|
|
keep_language_flags: True if language switching flags should be kept
|
|
no_stress: True if stress characters should be removed
|
|
|
|
Returns:
|
|
ipa - string of IPA phonemes
|
|
"""
|
|
self._maybe_init()
|
|
|
|
voice = voice or self.default_voice
|
|
|
|
if (voice is not None) and (voice != self.current_voice):
|
|
self.current_voice = voice
|
|
voice_bytes = voice.encode("utf-8")
|
|
result = self.lib_espeak.espeak_SetVoiceByName(voice_bytes)
|
|
assert result == Phonemizer.EE_OK, f"Failed to set voice to {voice}"
|
|
|
|
missing_breakers = []
|
|
if keep_clause_breakers and self.clause_breakers:
|
|
missing_breakers = [c for c in text if c in self.clause_breakers]
|
|
|
|
# Create in-memory file for phoneme trace.
|
|
# espeak_TextToPhonemes segfaults no matter what I do, so this is the back-up.
|
|
phonemes_buffer = ctypes.c_char_p()
|
|
phonemes_size = ctypes.c_size_t()
|
|
phonemes_file = self.libc.open_memstream(
|
|
ctypes.byref(phonemes_buffer), ctypes.byref(phonemes_size)
|
|
)
|
|
|
|
try:
|
|
phoneme_flags = Phonemizer.espeakPHONEMES_IPA
|
|
if phoneme_separator:
|
|
phoneme_flags = phoneme_flags | (ord(phoneme_separator) << 8)
|
|
|
|
self.lib_espeak.espeak_SetPhonemeTrace(phoneme_flags, phonemes_file)
|
|
|
|
text_bytes = text.encode("utf-8")
|
|
self.lib_espeak.espeak_Synth(
|
|
text_bytes,
|
|
0, # buflength (unused in AUDIO_OUTPUT_SYNCHRONOUS mode)
|
|
0, # position
|
|
0, # position_type
|
|
0, # end_position (no end position)
|
|
Phonemizer.espeakCHARS_AUTO | Phonemizer.espeakPHONEMES,
|
|
None, # unique_speaker,
|
|
None, # user_data,
|
|
)
|
|
self.libc.fflush(phonemes_file)
|
|
|
|
phoneme_lines = ctypes.string_at(phonemes_buffer).decode().splitlines()
|
|
|
|
if not keep_language_flags:
|
|
# Remove language switching flags, e.g. (en)
|
|
phoneme_lines = [
|
|
Phonemizer.LANG_SWITCH_FLAG.sub("", line) for line in phoneme_lines
|
|
]
|
|
|
|
if word_separator != " ":
|
|
# Split/re-join words
|
|
for line_idx in range(len(phoneme_lines)):
|
|
phoneme_lines[line_idx] = word_separator.join(
|
|
phoneme_lines[line_idx].split()
|
|
)
|
|
|
|
# Re-insert clause breakers
|
|
if missing_breakers:
|
|
# pylint: disable=consider-using-enumerate
|
|
for line_idx in range(len(phoneme_lines)):
|
|
if line_idx < len(missing_breakers):
|
|
phoneme_lines[line_idx] += (
|
|
punctuation_separator + missing_breakers[line_idx]
|
|
)
|
|
|
|
phonemes_str = word_separator.join(line.strip() for line in phoneme_lines)
|
|
|
|
if no_stress:
|
|
# Remove primary/secondary stress markers
|
|
phonemes_str = Phonemizer.STRESS_PATTERN.sub("", phonemes_str)
|
|
|
|
# Clean up multiple phoneme separators
|
|
if phoneme_separator:
|
|
phonemes_str = re.sub(
|
|
"[" + re.escape(phoneme_separator) + "]+",
|
|
phoneme_separator,
|
|
phonemes_str,
|
|
)
|
|
|
|
return phonemes_str
|
|
finally:
|
|
self.libc.fclose(phonemes_file)
|
|
|
|
def _maybe_init(self):
|
|
if self.libc and self.lib_espeak:
|
|
# Already initialized
|
|
return
|
|
|
|
self.libc = ctypes.cdll.LoadLibrary("libc.so.6")
|
|
self.libc.open_memstream.restype = ctypes.POINTER(ctypes.c_char)
|
|
|
|
try:
|
|
self.lib_espeak = ctypes.cdll.LoadLibrary("libespeak-ng.so")
|
|
except OSError:
|
|
# Try .so.1
|
|
self.lib_espeak = ctypes.cdll.LoadLibrary("libespeak-ng.so.1")
|
|
|
|
sample_rate = self.lib_espeak.espeak_Initialize(
|
|
Phonemizer.AUDIO_OUTPUT_SYNCHRONOUS, 0, None, 0
|
|
)
|
|
assert sample_rate > 0, "Failed to initialize libespeak-ng"
|