Add Farsi/Persian support with hazm

This commit is contained in:
Michael Hansen 2022-04-19 10:45:41 -04:00
commit 069a8df398
6 changed files with 299 additions and 19 deletions

View file

@ -75,17 +75,23 @@ RUN .venv/bin/pyinstaller \
--collect-data "gruut_lang_ru" \
--hidden-import "gruut_lang_sw" \
--collect-data "gruut_lang_sw" \
--hidden-import "gruut_lang_fa" \
--collect-data "gruut_lang_fa" \
--collect-data 'espeak_phonemizer' \
--collect-data 'phonemes2ids' \
--hidden-import 'swagger_ui' \
--hidden-import 'epitran' \
--hidden-import 'epitran' \
--collect-data 'panphon' \
--hidden-import 'hazm' \
--collect-data 'hazm' \
--collect-data 'panphon' \
--collect-data 'mimic3_tts' \
--collect-data 'mimic3_http' \
pyinstaller/mimic3.py
# Have to manually copy these over for some reason
RUN find .venv -name 'wapiti' -type d -exec cp -R {} dist/mimic3/ \;
RUN find .venv -name 'libwapiti.*.so' -type f -exec cp {} dist/mimic3/ \;
# Clean up unused lexicons
RUN find dist/mimic3/ -wholename '*/gruut_lang_*/espeak' -type d | \
while read -r espeak_dir; do \

View file

@ -1 +1 @@
0.1.5
0.1.6

View file

@ -329,7 +329,20 @@ class Mimic3Voice(metaclass=ABCMeta):
if config.phonemizer == Phonemizer.ESPEAK:
# Phonemes from eSpeak-ng: https://github.com/espeak-ng/espeak-ng
return EspeakVoice(
voice_class = EspeakVoice
if config.text_language == "fa":
try:
# Check if hazm is available
# https://github.com/sobhe/hazm
import hazm # noqa: F401
voice_class = HazmEspeakVoice
except ImportError:
_LOGGER.warning("hazm is highly recommended for language 'fa'")
_LOGGER.warning("pip install 'hazm>=0.7.0'")
return voice_class(
config=config,
onnx_model=onnx_model,
phoneme_to_id=phoneme_to_id,
@ -576,6 +589,109 @@ class EspeakVoice(Mimic3Voice):
return language.strip().lower().replace("_", "-")
class HazmEspeakVoice(EspeakVoice):
"""Persian espeak-ng voice that uses hazm (https://github.com/sobhe/hazm) for pre-processing"""
def __init__(self, *args, **kwargs):
import gruut_lang_fa
import hazm
super().__init__(*args, **kwargs)
self._normalizer = hazm.Normalizer()
self._sent_tokenizer = hazm.SentenceTokenizer()
self._word_tokenizer = hazm.WordTokenizer()
# Load part of speech tagger from gruut[fa]
self._tagger = hazm.POSTagger(
model=str(gruut_lang_fa.get_lang_dir() / "pos" / "postagger.model")
)
def text_to_phonemes(
self, text: str, text_language: typing.Optional[str] = None
) -> TEXT_TO_PHONEMES_TYPE:
phoneme_separator = ""
word_separator = self.config.phonemes.word_separator
text_language = text_language or self.config.text_language or DEFAULT_LANGUAGE
voice = self._language_to_voice(text_language)
# Normalize with hazm
sentences = self._preprocess_text(text)
for sentence in sentences:
sent_text = " ".join(sentence)
sent_phoneme_str = self._phonemizer.phonemize(
sent_text,
voice=voice,
keep_clause_breakers=True,
phoneme_separator=phoneme_separator,
word_separator=word_separator,
punctuation_separator=phoneme_separator,
)
sent_word_phonemes = [
list(IPA.graphemes(wp_str))
for wp_str in sent_phoneme_str.split(word_separator)
]
yield sent_word_phonemes, BreakType.UTTERANCE
def word_to_phonemes(
self,
word_text: str,
word_role: typing.Optional[str] = None,
text_language: typing.Optional[str] = None,
) -> typing.List[PHONEME_TYPE]:
word_text = self._fix_words([word_text])[0]
return super().word_to_phonemes(
word_text, word_role=word_role, text_language=text_language
)
def say_as_to_phonemes(
self,
text: str,
interpret_as: str,
say_format: typing.Optional[str] = None,
text_language: typing.Optional[str] = None,
) -> WORD_PHONEMES_TYPE:
sentences = self._preprocess_text(text)
text = " ".join(
" ".join(word_text for word_text in words) for words in sentences
)
return super().say_as_to_phonemes(
text, interpret_as, say_format=say_format, text_language=text_language
)
def _preprocess_text(self, text: str) -> typing.List[typing.List[str]]:
"""Split/normalize text into sentences/words with hazm"""
text = self._normalizer.normalize(text)
processed_sentences = []
for sentence in self._sent_tokenizer.tokenize(text):
words = self._word_tokenizer.tokenize(sentence)
processed_words = self._fix_words(words)
processed_sentences.append(processed_words)
return processed_sentences
def _fix_words(self, words: typing.List[str]) -> typing.List[str]:
fixed_words = []
for word, pos in self._tagger.tag(words):
if pos[-1] == "e":
if word[-1] != "ِ":
if (word[-1] == "ه") and (word[-2] != "ا"):
word += "‌ی"
word += "ِ"
fixed_words.append(word)
return fixed_words
# -----------------------------------------------------------------------------

View file

@ -141,25 +141,13 @@
"size_bytes": 18,
"sha256_sum": "8197ffe96f3b6772797357e007d63cde409573a0bd3fe174489e01a5faa95553"
},
"_generator-opt.onnx": {
"size_bytes": 62674344,
"sha256_sum": "4244f801bd0cd64e4079295e116ceb830de1f6575764b59f8e5d0e0e7fd14dc7"
},
"config.json": {
"size_bytes": 3434,
"sha256_sum": "1fdaa1124e02cc177eb776fbc6e08c838b56bd2e86c82d8d7fe434d9337806b0"
},
"generator.fp16.onnx": {
"size_bytes": 31564712,
"sha256_sum": "106188c8b75f137f490213c350943fdc28b55e892c25ab1743277b2104c2c1a1"
},
"generator.quant.onnx": {
"size_bytes": 18191527,
"sha256_sum": "38832936c39fc7cc46cb042f4e6f3817f69d3c5861bdd93c51a385c1ae60456d"
},
"generator.squant.onnx": {
"size_bytes": 18191458,
"sha256_sum": "acc2bba2f1a866896192fc1487dc36d7d598640d8c66c83ad6c8f10d7f002811"
"generator.onnx": {
"size_bytes": 62792219,
"sha256_sum": "0b5a323500ebd022351db12da2b3aab8cdd47d0826d173e780a58b93604618c9"
},
"phoneme_map.txt": {
"size_bytes": 15,
@ -276,6 +264,56 @@
"speakers": [],
"properties": {}
},
"en_US/m-ailabs_low": {
"files": {
"ALIASES": {
"size_bytes": 15,
"sha256_sum": "7bb223e3ab3fd2fb2f376bf6648bd9ffb60c1f94ebed81b6ab224b04ade583a2"
},
"LICENSE": {
"size_bytes": 1372,
"sha256_sum": "fdd78a909fb9384d869363522b967557bc9e28e5b65874921f24e48cbb82f38c"
},
"README.md": {
"size_bytes": 204,
"sha256_sum": "bbfa383da209e57d7432ff5a3f6fc2eb4f751f469fda7fa556fef361646ee814"
},
"SOURCE": {
"size_bytes": 61,
"sha256_sum": "841520f6a8cc616e307a92552355691f8c3087fadda2e9b7a03a7863b2d0cf6a"
},
"config.json": {
"size_bytes": 3656,
"sha256_sum": "4cc13cff911251e35c6b9e9c62d3239c59edc75a144753aaed6606adf23fc36f"
},
"generator.onnx": {
"size_bytes": 76329055,
"sha256_sum": "e92b3bb30462fee4bb7bfed8b341995f6665c1ffcd8d0489acfa42229647f036"
},
"phoneme_map.txt": {
"size_bytes": 26,
"sha256_sum": "be897ed405d294ec18acb94cd2d33fe262e90b5bbafc3f01cbd75694027a0856"
},
"phonemes.txt": {
"size_bytes": 263,
"sha256_sum": "8f9c3e6ced14d7fc5426e4e1bc7f7cc1037a20a645ca34110abcb76148fa8bfd"
},
"speaker_map.csv": {
"size_bytes": 71,
"sha256_sum": "2883321465b6aeab0b2bad54d9aef76cd7c767b6f7e704662ec1589f22df2524"
},
"speakers.txt": {
"size_bytes": 38,
"sha256_sum": "66aecfe5f8e8cb7e223e78b3e8cae7a7f2a93e6ae09083d3211e2d1f645b48fb"
}
},
"speakers": [
"elliot_miller",
"judy_bieber",
"mary_ann"
],
"properties": {}
},
"en_US/vctk_low": {
"files": {
"ALIASES": {
@ -516,6 +554,36 @@
],
"properties": {}
},
"fa/haaniye_low": {
"files": {
"LICENSE": {
"size_bytes": 5,
"sha256_sum": "958246282e394a96727515ee6d834073ea09314be61a5730d03bf6d540ae0c26"
},
"README.md": {
"size_bytes": 145,
"sha256_sum": "b7a0d433277c03526bfd3d116f0b8ba25b8f41baa8abf744291c8847f89d1fdd"
},
"SOURCE": {
"size_bytes": 4,
"sha256_sum": "76f602b1dfc2d548fa1f5eddf29315c947bec31f71c95d49653927204486be39"
},
"config.json": {
"size_bytes": 3667,
"sha256_sum": "fd6d675532c06e014080df6abffc870b3c299dd31e28515044858072d2e6741a"
},
"generator.onnx": {
"size_bytes": 62783767,
"sha256_sum": "84ab78157850c4f80a884f730d88f3e091a69c98772645da87b418fc325e23e9"
},
"phonemes.txt": {
"size_bytes": 183,
"sha256_sum": "85753b8b47cdb68d817c3c2fd3a47ebe85abcd3757d5348e86204c30595a4fa2"
}
},
"speakers": [],
"properties": {}
},
"fi_FI/harri-tapani-ylilammi_low": {
"files": {
"ALIASES": {
@ -712,6 +780,88 @@
"speakers": [],
"properties": {}
},
"it_IT/mls_low": {
"files": {
"ALIASES": {
"size_bytes": 10,
"sha256_sum": "a9e36d6496c023b6a062cbad69f72b3d97ac3f460fcfde05e58c8b1fee563005"
},
"LICENSE": {
"size_bytes": 11,
"sha256_sum": "e133cf7dafe75a1763bd79fd3dc16859d9bfef17b5687c37eab1c88b7f5d12ca"
},
"README.md": {
"size_bytes": 150,
"sha256_sum": "7a8f7a21659afc8efbca65dee85bbfc29318b7b787c9ff6194b0d262f8a56214"
},
"SOURCE": {
"size_bytes": 27,
"sha256_sum": "e381746e5212214dc753e4dd5b74e31da0271cfb0f529b2aaa9b721da5a0a84f"
},
"config.json": {
"size_bytes": 3634,
"sha256_sum": "36158d2347a4cf896125c5b4adf4444f9a21e2b1184b358d36e1f2964a62dc09"
},
"generator.onnx": {
"size_bytes": 76396641,
"sha256_sum": "98125e13d9809353c723d5800a717b57ead070f76875dc1024cd60129e63ac69"
},
"phonemes.txt": {
"size_bytes": 210,
"sha256_sum": "282837161676bffa5b304cbb878eace1c8da670a46e08e8e800515f924ecfde3"
},
"speaker_map.csv": {
"size_bytes": 495,
"sha256_sum": "4dd7a8a3c9e1b85c60948c22b4eab98408125269601d5b2529ad1dea7b90e36e"
},
"speakers.txt": {
"size_bytes": 232,
"sha256_sum": "bbbeceb499b9b59b704de555aba3202d20afa1d3a6c854112e3f229b8fef52f3"
}
},
"speakers": [
"1595",
"4974",
"4998",
"6807",
"1989",
"2033",
"2019",
"659",
"4649",
"9772",
"1725",
"10446",
"6348",
"6001",
"9185",
"8842",
"8828",
"12428",
"8181",
"7440",
"8207",
"277",
"5421",
"12804",
"4705",
"7936",
"844",
"6299",
"644",
"8384",
"1157",
"7444",
"643",
"4971",
"4975",
"6744",
"8461",
"7405",
"5010"
],
"properties": {}
},
"it_IT/riccardo-fasol_low": {
"files": {
"ALIASES": {

View file

@ -6,6 +6,12 @@ ignore_missing_imports = True
[mypy-epitran.*]
ignore_missing_imports = True
[mypy-gruut_lang_fa.*]
ignore_missing_imports = True
[mypy-hazm.*]
ignore_missing_imports = True
[mypy-onnxruntime.*]
ignore_missing_imports = True

View file

@ -50,6 +50,7 @@ extras = {}
for lang in [
"de",
"es",
"fa",
"fr",
"it",
"nl",
@ -58,6 +59,7 @@ for lang in [
]:
extras[f"gruut[{lang}]"] = [lang]
# Add "all" tag
for tags in extras.values():
tags.append("all")