Add <prosody rate="...">
This commit is contained in:
parent
bac647841f
commit
a3a159d2f6
8 changed files with 101 additions and 3 deletions
|
|
@ -242,7 +242,7 @@ For example:
|
||||||
<break time="3s" />
|
<break time="3s" />
|
||||||
<voice name="en_US/cmu-arctic_low#slt">
|
<voice name="en_US/cmu-arctic_low#slt">
|
||||||
<s>
|
<s>
|
||||||
<prosody volume="soft">
|
<prosody volume="soft" rate="150%">
|
||||||
This is a <say-as interpret-as="number" format="ordinal">2</say-as> voice.
|
This is a <say-as interpret-as="number" format="ordinal">2</say-as> voice.
|
||||||
</prosody>
|
</prosody>
|
||||||
</s>
|
</s>
|
||||||
|
|
|
||||||
|
|
@ -17,6 +17,7 @@
|
||||||
import argparse
|
import argparse
|
||||||
import asyncio
|
import asyncio
|
||||||
import dataclasses
|
import dataclasses
|
||||||
|
import json
|
||||||
import logging
|
import logging
|
||||||
import typing
|
import typing
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
@ -119,6 +120,17 @@ def get_app(args: argparse.Namespace, request_queue: Queue, temp_dir: str):
|
||||||
def _to_bool(s: str) -> bool:
|
def _to_bool(s: str) -> bool:
|
||||||
return s.strip().lower() in {"true", "1", "yes", "on"}
|
return s.strip().lower() in {"true", "1", "yes", "on"}
|
||||||
|
|
||||||
|
class VoiceEncoder(json.JSONEncoder):
|
||||||
|
"""Encode a voice to JSON"""
|
||||||
|
|
||||||
|
def default(self, o):
|
||||||
|
if isinstance(o, set):
|
||||||
|
return list(o)
|
||||||
|
|
||||||
|
return json.JSONEncoder.default(self, o)
|
||||||
|
|
||||||
|
app.json_encoder = VoiceEncoder # type: ignore
|
||||||
|
|
||||||
@app.route("/img/<path:filename>", methods=["GET"])
|
@app.route("/img/<path:filename>", methods=["GET"])
|
||||||
async def img(filename) -> Response:
|
async def img(filename) -> Response:
|
||||||
"""Image static endpoint."""
|
"""Image static endpoint."""
|
||||||
|
|
|
||||||
|
|
@ -178,6 +178,10 @@ A subset of [SSML](https://www.w3.org/TR/speech-synthesis11/) (Speech Synthesis
|
||||||
* number in [0, 100] - 0 is silent, 100 is loudest (default)
|
* number in [0, 100] - 0 is silent, 100 is loudest (default)
|
||||||
* +X, -X, +X%, -X% - absolute/percent offset from current volume
|
* +X, -X, +X%, -X% - absolute/percent offset from current volume
|
||||||
* one of "default", "silent", "x-loud", "loud", "medium", "soft", "x-soft"
|
* one of "default", "silent", "x-loud", "loud", "medium", "soft", "x-soft"
|
||||||
|
* `rate` - speaking rate
|
||||||
|
* number - 1 is default rate, < 1 is slower, > 1 is faster
|
||||||
|
* X% - 100% is default rate, 50% is half speed, 200% is twice as fast
|
||||||
|
* one of "default", "x-fast", "fast", "medium", "slow", "x-slow"
|
||||||
* `<say-as interpret-as="">` - force interpretation of inner text
|
* `<say-as interpret-as="">` - force interpretation of inner text
|
||||||
* `interpret-as` one of "spell-out", "date", "number", "time", or "currency"
|
* `interpret-as` one of "spell-out", "date", "number", "time", or "currency"
|
||||||
* `format` - way to format text depending on `interpret-as`
|
* `format` - way to format text depending on `interpret-as`
|
||||||
|
|
|
||||||
|
|
@ -25,3 +25,4 @@ DEFAULT_VOICES_URL_FORMAT = (
|
||||||
DEFAULT_VOICES_DOWNLOAD_DIR = Path(XDG().XDG_DATA_HOME) / "mimic3" / "voices"
|
DEFAULT_VOICES_DOWNLOAD_DIR = Path(XDG().XDG_DATA_HOME) / "mimic3" / "voices"
|
||||||
|
|
||||||
DEFAULT_VOLUME = 100.0
|
DEFAULT_VOLUME = 100.0
|
||||||
|
DEFAULT_RATE = 1.0
|
||||||
|
|
|
||||||
|
|
@ -40,6 +40,7 @@ from ._resources import _VOICES
|
||||||
from .config import TrainingConfig
|
from .config import TrainingConfig
|
||||||
from .const import (
|
from .const import (
|
||||||
DEFAULT_LANGUAGE,
|
DEFAULT_LANGUAGE,
|
||||||
|
DEFAULT_RATE,
|
||||||
DEFAULT_VOICE,
|
DEFAULT_VOICE,
|
||||||
DEFAULT_VOICES_DOWNLOAD_DIR,
|
DEFAULT_VOICES_DOWNLOAD_DIR,
|
||||||
DEFAULT_VOICES_URL_FORMAT,
|
DEFAULT_VOICES_URL_FORMAT,
|
||||||
|
|
@ -113,6 +114,9 @@ class Mimic3Settings:
|
||||||
volume: float = DEFAULT_VOLUME
|
volume: float = DEFAULT_VOLUME
|
||||||
"""Voice volume in [0, 100]"""
|
"""Voice volume in [0, 100]"""
|
||||||
|
|
||||||
|
rate: float = DEFAULT_RATE
|
||||||
|
"""Voice speaking rate (< 1 is slower, > 1 is faster)"""
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class Mimic3Phonemes:
|
class Mimic3Phonemes:
|
||||||
|
|
@ -318,6 +322,14 @@ class Mimic3TextToSpeechSystem(TextToSpeechSystem):
|
||||||
def volume(self, new_volume: float):
|
def volume(self, new_volume: float):
|
||||||
self.settings.volume = max(0, min(100, new_volume))
|
self.settings.volume = max(0, min(100, new_volume))
|
||||||
|
|
||||||
|
@property
|
||||||
|
def rate(self) -> float:
|
||||||
|
return self.settings.rate
|
||||||
|
|
||||||
|
@rate.setter
|
||||||
|
def rate(self, new_rate: float):
|
||||||
|
self.settings.rate = new_rate
|
||||||
|
|
||||||
def begin_utterance(self):
|
def begin_utterance(self):
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
@ -456,6 +468,7 @@ class Mimic3TextToSpeechSystem(TextToSpeechSystem):
|
||||||
length_scale=settings.length_scale,
|
length_scale=settings.length_scale,
|
||||||
noise_scale=settings.noise_scale,
|
noise_scale=settings.noise_scale,
|
||||||
noise_w=settings.noise_w,
|
noise_w=settings.noise_w,
|
||||||
|
rate=settings.rate,
|
||||||
)
|
)
|
||||||
|
|
||||||
audio_bytes = audio.tobytes()
|
audio_bytes = audio.tobytes()
|
||||||
|
|
|
||||||
|
|
@ -32,8 +32,9 @@ import onnxruntime
|
||||||
import phonemes2ids
|
import phonemes2ids
|
||||||
from gruut_ipa import IPA
|
from gruut_ipa import IPA
|
||||||
|
|
||||||
from mimic3_tts.config import Phonemizer, TrainingConfig
|
from .config import Phonemizer, TrainingConfig
|
||||||
from mimic3_tts.utils import audio_float_to_int16
|
from .const import DEFAULT_RATE
|
||||||
|
from .utils import audio_float_to_int16
|
||||||
|
|
||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
@ -159,11 +160,16 @@ class Mimic3Voice(metaclass=ABCMeta):
|
||||||
length_scale: typing.Optional[float] = None,
|
length_scale: typing.Optional[float] = None,
|
||||||
noise_scale: typing.Optional[float] = None,
|
noise_scale: typing.Optional[float] = None,
|
||||||
noise_w: typing.Optional[float] = None,
|
noise_w: typing.Optional[float] = None,
|
||||||
|
rate: float = DEFAULT_RATE,
|
||||||
) -> np.ndarray:
|
) -> np.ndarray:
|
||||||
"""Synthesize audio from phoneme ids usng Onnx voice model (see generator.onnx)"""
|
"""Synthesize audio from phoneme ids usng Onnx voice model (see generator.onnx)"""
|
||||||
if length_scale is None:
|
if length_scale is None:
|
||||||
length_scale = self.config.inference.length_scale
|
length_scale = self.config.inference.length_scale
|
||||||
|
|
||||||
|
# Scale length by rate
|
||||||
|
if rate > 0:
|
||||||
|
length_scale /= rate
|
||||||
|
|
||||||
if noise_scale is None:
|
if noise_scale is None:
|
||||||
noise_scale = self.config.inference.noise_scale
|
noise_scale = self.config.inference.noise_scale
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -218,6 +218,15 @@ class TextToSpeechSystem(AbstractContextManager, metaclass=ABCMeta):
|
||||||
def volume(self, new_volume: float):
|
def volume(self, new_volume: float):
|
||||||
"""Set the current volume in [0, 100]"""
|
"""Set the current volume in [0, 100]"""
|
||||||
|
|
||||||
|
@property
|
||||||
|
@abstractmethod
|
||||||
|
def rate(self) -> float:
|
||||||
|
"""Get the current speaking rate"""
|
||||||
|
|
||||||
|
@rate.setter
|
||||||
|
def rate(self, new_rate: float):
|
||||||
|
"""Set the current speaker rate"""
|
||||||
|
|
||||||
def shutdown(self):
|
def shutdown(self):
|
||||||
"""Called by the host program when the text to speech system should be stopped"""
|
"""Called by the host program when the text to speech system should be stopped"""
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -63,6 +63,7 @@ class ParsingState(int, enum.Enum):
|
||||||
|
|
||||||
|
|
||||||
_DEFAULT_VOLUME: float = 100.0
|
_DEFAULT_VOLUME: float = 100.0
|
||||||
|
_DEFAULT_RATE: float = 1.0
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
|
|
@ -70,6 +71,10 @@ class ProsodyState:
|
||||||
"""Current prosody settings"""
|
"""Current prosody settings"""
|
||||||
|
|
||||||
volume: float = _DEFAULT_VOLUME
|
volume: float = _DEFAULT_VOLUME
|
||||||
|
"""Currrent volume setting in [0, 100]"""
|
||||||
|
|
||||||
|
rate: float = _DEFAULT_RATE
|
||||||
|
"""Current rate setting (< 1 is slower, > 1 is faster)"""
|
||||||
|
|
||||||
|
|
||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
|
|
@ -84,12 +89,29 @@ _DEFAULT_VOLUME_MAP = {
|
||||||
"silent": 0.0,
|
"silent": 0.0,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
_DEFAULT_RATE_MAP = {
|
||||||
|
"default": _DEFAULT_RATE,
|
||||||
|
"x-fast": _DEFAULT_RATE * 3,
|
||||||
|
"fast": _DEFAULT_RATE * 2,
|
||||||
|
"medium": _DEFAULT_RATE,
|
||||||
|
"slow": _DEFAULT_RATE * 0.5,
|
||||||
|
"x-slow": _DEFAULT_RATE * 0.25,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class SSMLSettings:
|
class SSMLSettings:
|
||||||
|
"""Settings for SSML named constants"""
|
||||||
|
|
||||||
volume_map: typing.Mapping[str, float] = field(
|
volume_map: typing.Mapping[str, float] = field(
|
||||||
default_factory=lambda: _DEFAULT_VOLUME_MAP
|
default_factory=lambda: _DEFAULT_VOLUME_MAP
|
||||||
)
|
)
|
||||||
|
"""Volume named constant to volume level"""
|
||||||
|
|
||||||
|
rate_map: typing.Mapping[str, float] = field(
|
||||||
|
default_factory=lambda: _DEFAULT_RATE_MAP
|
||||||
|
)
|
||||||
|
"""Rate named constant to speaking rate"""
|
||||||
|
|
||||||
|
|
||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
|
|
@ -449,10 +471,15 @@ class SSMLSpeaker:
|
||||||
volume_str, current_volume=self._prosody.volume
|
volume_str, current_volume=self._prosody.volume
|
||||||
)
|
)
|
||||||
|
|
||||||
|
rate_str = attrib_no_namespace(elem, "rate")
|
||||||
|
if rate_str is not None:
|
||||||
|
new_prosody.rate = self._parse_rate(rate_str)
|
||||||
|
|
||||||
LOG.debug("prosody: %s", new_prosody)
|
LOG.debug("prosody: %s", new_prosody)
|
||||||
self._push_prosody(new_prosody)
|
self._push_prosody(new_prosody)
|
||||||
|
|
||||||
self.tts.volume = new_prosody.volume
|
self.tts.volume = new_prosody.volume
|
||||||
|
self.tts.rate = new_prosody.rate
|
||||||
|
|
||||||
def _handle_end_prosody(self):
|
def _handle_end_prosody(self):
|
||||||
"""Handle </prosody>"""
|
"""Handle </prosody>"""
|
||||||
|
|
@ -607,6 +634,32 @@ class SSMLSpeaker:
|
||||||
|
|
||||||
return max(0, min(_DEFAULT_VOLUME, volume))
|
return max(0, min(_DEFAULT_VOLUME, volume))
|
||||||
|
|
||||||
|
def _parse_rate(self, rate_str: str) -> float:
|
||||||
|
"""Parse SSML rate from <prosody> into float"""
|
||||||
|
rate = _DEFAULT_RATE
|
||||||
|
rate_str = rate_str.strip().lower()
|
||||||
|
|
||||||
|
maybe_rate = self.settings.rate_map.get(rate_str)
|
||||||
|
if maybe_rate is not None:
|
||||||
|
rate = maybe_rate
|
||||||
|
elif rate_str:
|
||||||
|
is_percent = False
|
||||||
|
|
||||||
|
if rate_str[-1] == "%":
|
||||||
|
is_percent = True
|
||||||
|
rate_str = rate_str[:-1]
|
||||||
|
|
||||||
|
rate_value = float(rate_str)
|
||||||
|
|
||||||
|
if is_percent:
|
||||||
|
# 50% = 0.5
|
||||||
|
rate = rate_value / 100.0
|
||||||
|
else:
|
||||||
|
# Absolute value
|
||||||
|
rate = rate_value
|
||||||
|
|
||||||
|
return rate
|
||||||
|
|
||||||
|
|
||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue