Add <prosody rate="...">
This commit is contained in:
parent
bac647841f
commit
a3a159d2f6
8 changed files with 101 additions and 3 deletions
|
|
@ -242,7 +242,7 @@ For example:
|
|||
<break time="3s" />
|
||||
<voice name="en_US/cmu-arctic_low#slt">
|
||||
<s>
|
||||
<prosody volume="soft">
|
||||
<prosody volume="soft" rate="150%">
|
||||
This is a <say-as interpret-as="number" format="ordinal">2</say-as> voice.
|
||||
</prosody>
|
||||
</s>
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@
|
|||
import argparse
|
||||
import asyncio
|
||||
import dataclasses
|
||||
import json
|
||||
import logging
|
||||
import typing
|
||||
from pathlib import Path
|
||||
|
|
@ -119,6 +120,17 @@ def get_app(args: argparse.Namespace, request_queue: Queue, temp_dir: str):
|
|||
def _to_bool(s: str) -> bool:
|
||||
return s.strip().lower() in {"true", "1", "yes", "on"}
|
||||
|
||||
class VoiceEncoder(json.JSONEncoder):
|
||||
"""Encode a voice to JSON"""
|
||||
|
||||
def default(self, o):
|
||||
if isinstance(o, set):
|
||||
return list(o)
|
||||
|
||||
return json.JSONEncoder.default(self, o)
|
||||
|
||||
app.json_encoder = VoiceEncoder # type: ignore
|
||||
|
||||
@app.route("/img/<path:filename>", methods=["GET"])
|
||||
async def img(filename) -> Response:
|
||||
"""Image static endpoint."""
|
||||
|
|
|
|||
|
|
@ -178,6 +178,10 @@ A subset of [SSML](https://www.w3.org/TR/speech-synthesis11/) (Speech Synthesis
|
|||
* number in [0, 100] - 0 is silent, 100 is loudest (default)
|
||||
* +X, -X, +X%, -X% - absolute/percent offset from current volume
|
||||
* one of "default", "silent", "x-loud", "loud", "medium", "soft", "x-soft"
|
||||
* `rate` - speaking rate
|
||||
* number - 1 is default rate, < 1 is slower, > 1 is faster
|
||||
* X% - 100% is default rate, 50% is half speed, 200% is twice as fast
|
||||
* one of "default", "x-fast", "fast", "medium", "slow", "x-slow"
|
||||
* `<say-as interpret-as="">` - force interpretation of inner text
|
||||
* `interpret-as` one of "spell-out", "date", "number", "time", or "currency"
|
||||
* `format` - way to format text depending on `interpret-as`
|
||||
|
|
|
|||
|
|
@ -25,3 +25,4 @@ DEFAULT_VOICES_URL_FORMAT = (
|
|||
DEFAULT_VOICES_DOWNLOAD_DIR = Path(XDG().XDG_DATA_HOME) / "mimic3" / "voices"
|
||||
|
||||
DEFAULT_VOLUME = 100.0
|
||||
DEFAULT_RATE = 1.0
|
||||
|
|
|
|||
|
|
@ -40,6 +40,7 @@ from ._resources import _VOICES
|
|||
from .config import TrainingConfig
|
||||
from .const import (
|
||||
DEFAULT_LANGUAGE,
|
||||
DEFAULT_RATE,
|
||||
DEFAULT_VOICE,
|
||||
DEFAULT_VOICES_DOWNLOAD_DIR,
|
||||
DEFAULT_VOICES_URL_FORMAT,
|
||||
|
|
@ -113,6 +114,9 @@ class Mimic3Settings:
|
|||
volume: float = DEFAULT_VOLUME
|
||||
"""Voice volume in [0, 100]"""
|
||||
|
||||
rate: float = DEFAULT_RATE
|
||||
"""Voice speaking rate (< 1 is slower, > 1 is faster)"""
|
||||
|
||||
|
||||
@dataclass
|
||||
class Mimic3Phonemes:
|
||||
|
|
@ -318,6 +322,14 @@ class Mimic3TextToSpeechSystem(TextToSpeechSystem):
|
|||
def volume(self, new_volume: float):
|
||||
self.settings.volume = max(0, min(100, new_volume))
|
||||
|
||||
@property
|
||||
def rate(self) -> float:
|
||||
return self.settings.rate
|
||||
|
||||
@rate.setter
|
||||
def rate(self, new_rate: float):
|
||||
self.settings.rate = new_rate
|
||||
|
||||
def begin_utterance(self):
|
||||
pass
|
||||
|
||||
|
|
@ -456,6 +468,7 @@ class Mimic3TextToSpeechSystem(TextToSpeechSystem):
|
|||
length_scale=settings.length_scale,
|
||||
noise_scale=settings.noise_scale,
|
||||
noise_w=settings.noise_w,
|
||||
rate=settings.rate,
|
||||
)
|
||||
|
||||
audio_bytes = audio.tobytes()
|
||||
|
|
|
|||
|
|
@ -32,8 +32,9 @@ import onnxruntime
|
|||
import phonemes2ids
|
||||
from gruut_ipa import IPA
|
||||
|
||||
from mimic3_tts.config import Phonemizer, TrainingConfig
|
||||
from mimic3_tts.utils import audio_float_to_int16
|
||||
from .config import Phonemizer, TrainingConfig
|
||||
from .const import DEFAULT_RATE
|
||||
from .utils import audio_float_to_int16
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
|
@ -159,11 +160,16 @@ class Mimic3Voice(metaclass=ABCMeta):
|
|||
length_scale: typing.Optional[float] = None,
|
||||
noise_scale: typing.Optional[float] = None,
|
||||
noise_w: typing.Optional[float] = None,
|
||||
rate: float = DEFAULT_RATE,
|
||||
) -> np.ndarray:
|
||||
"""Synthesize audio from phoneme ids usng Onnx voice model (see generator.onnx)"""
|
||||
if length_scale is None:
|
||||
length_scale = self.config.inference.length_scale
|
||||
|
||||
# Scale length by rate
|
||||
if rate > 0:
|
||||
length_scale /= rate
|
||||
|
||||
if noise_scale is None:
|
||||
noise_scale = self.config.inference.noise_scale
|
||||
|
||||
|
|
|
|||
|
|
@ -218,6 +218,15 @@ class TextToSpeechSystem(AbstractContextManager, metaclass=ABCMeta):
|
|||
def volume(self, new_volume: float):
|
||||
"""Set the current volume in [0, 100]"""
|
||||
|
||||
@property
|
||||
@abstractmethod
|
||||
def rate(self) -> float:
|
||||
"""Get the current speaking rate"""
|
||||
|
||||
@rate.setter
|
||||
def rate(self, new_rate: float):
|
||||
"""Set the current speaker rate"""
|
||||
|
||||
def shutdown(self):
|
||||
"""Called by the host program when the text to speech system should be stopped"""
|
||||
|
||||
|
|
|
|||
|
|
@ -63,6 +63,7 @@ class ParsingState(int, enum.Enum):
|
|||
|
||||
|
||||
_DEFAULT_VOLUME: float = 100.0
|
||||
_DEFAULT_RATE: float = 1.0
|
||||
|
||||
|
||||
@dataclass
|
||||
|
|
@ -70,6 +71,10 @@ class ProsodyState:
|
|||
"""Current prosody settings"""
|
||||
|
||||
volume: float = _DEFAULT_VOLUME
|
||||
"""Currrent volume setting in [0, 100]"""
|
||||
|
||||
rate: float = _DEFAULT_RATE
|
||||
"""Current rate setting (< 1 is slower, > 1 is faster)"""
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
|
|
@ -84,12 +89,29 @@ _DEFAULT_VOLUME_MAP = {
|
|||
"silent": 0.0,
|
||||
}
|
||||
|
||||
_DEFAULT_RATE_MAP = {
|
||||
"default": _DEFAULT_RATE,
|
||||
"x-fast": _DEFAULT_RATE * 3,
|
||||
"fast": _DEFAULT_RATE * 2,
|
||||
"medium": _DEFAULT_RATE,
|
||||
"slow": _DEFAULT_RATE * 0.5,
|
||||
"x-slow": _DEFAULT_RATE * 0.25,
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class SSMLSettings:
|
||||
"""Settings for SSML named constants"""
|
||||
|
||||
volume_map: typing.Mapping[str, float] = field(
|
||||
default_factory=lambda: _DEFAULT_VOLUME_MAP
|
||||
)
|
||||
"""Volume named constant to volume level"""
|
||||
|
||||
rate_map: typing.Mapping[str, float] = field(
|
||||
default_factory=lambda: _DEFAULT_RATE_MAP
|
||||
)
|
||||
"""Rate named constant to speaking rate"""
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
|
|
@ -449,10 +471,15 @@ class SSMLSpeaker:
|
|||
volume_str, current_volume=self._prosody.volume
|
||||
)
|
||||
|
||||
rate_str = attrib_no_namespace(elem, "rate")
|
||||
if rate_str is not None:
|
||||
new_prosody.rate = self._parse_rate(rate_str)
|
||||
|
||||
LOG.debug("prosody: %s", new_prosody)
|
||||
self._push_prosody(new_prosody)
|
||||
|
||||
self.tts.volume = new_prosody.volume
|
||||
self.tts.rate = new_prosody.rate
|
||||
|
||||
def _handle_end_prosody(self):
|
||||
"""Handle </prosody>"""
|
||||
|
|
@ -607,6 +634,32 @@ class SSMLSpeaker:
|
|||
|
||||
return max(0, min(_DEFAULT_VOLUME, volume))
|
||||
|
||||
def _parse_rate(self, rate_str: str) -> float:
|
||||
"""Parse SSML rate from <prosody> into float"""
|
||||
rate = _DEFAULT_RATE
|
||||
rate_str = rate_str.strip().lower()
|
||||
|
||||
maybe_rate = self.settings.rate_map.get(rate_str)
|
||||
if maybe_rate is not None:
|
||||
rate = maybe_rate
|
||||
elif rate_str:
|
||||
is_percent = False
|
||||
|
||||
if rate_str[-1] == "%":
|
||||
is_percent = True
|
||||
rate_str = rate_str[:-1]
|
||||
|
||||
rate_value = float(rate_str)
|
||||
|
||||
if is_percent:
|
||||
# 50% = 0.5
|
||||
rate = rate_value / 100.0
|
||||
else:
|
||||
# Absolute value
|
||||
rate = rate_value
|
||||
|
||||
return rate
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue