Clean up opentts_abc

This commit is contained in:
Michael Hansen 2022-03-25 11:52:23 -04:00
commit bfdc26621e
4 changed files with 318 additions and 179 deletions

View file

@ -1,68 +1,107 @@
#!/usr/bin/env python3
"""Base classes for Open Text to Speech systems"""
import dataclasses
import io
import typing
import wave
from abc import ABCMeta, abstractmethod
from contextlib import AbstractContextManager
from copy import deepcopy
from dataclasses import dataclass
@dataclass
class Settings:
"""Current settings for TTS system"""
voice: typing.Optional[str] = None
"""Current voice key"""
language: typing.Optional[str] = None
"""Current language (e.g., en_US)"""
volume: typing.Optional[float] = None
"""Current speaking volume"""
rate: typing.Optional[float] = None
"""Current speaking rate"""
pitch: typing.Optional[float] = None
active_lexicons: typing.Optional[typing.Sequence[str]] = None
"""Current speaking pitch"""
other_settings: typing.Optional[typing.Mapping[str, typing.Any]] = None
"""Custom settings"""
@dataclass
class BaseToken(metaclass=ABCMeta):
"""Base class for spoken tokens"""
text: str
"""Text of the token"""
@dataclass
class Word(BaseToken):
"""Token representing a single word"""
role: typing.Optional[str] = None
"""Role of the word (typically part of speech)"""
@dataclass
class Phonemes(BaseToken):
"""Token representing a phonemized word"""
alphabet: typing.Optional[str] = None
"""Phoneme alphabet (e.g., ipa)"""
@dataclass
class SayAs(BaseToken):
"""Token representing a word or phrase that must be spoken a particular way"""
interpret_as: str
"""Implementation-dependent token interpretation (e.g., characters or digits)"""
format: typing.Optional[str] = None
"""Implementation-dependent token format (depends on interpret_as)"""
@dataclass
class _BaseResultDefaults:
"""Base class of results from TTS end_utterance"""
tag: typing.Optional[typing.Any] = None
"""Optional tag to associate with results"""
@dataclass
class BaseResult(metaclass=ABCMeta):
pass
"""Base class of results from TTS end_utterance"""
@dataclass
class _AudioResultBase:
"""Synthesized audio result"""
sample_rate_hz: int
"""Sample rate in Hertz (e.g., 22050)"""
sample_width_bytes: int
"""Sample width in bytes (e.g., 2)"""
num_channels: int
"""Number of audio channels (e.g., 1)"""
audio_bytes: bytes
"""Raw audio bytes (no header)"""
@dataclass
class AudioResult(BaseResult, _BaseResultDefaults, _AudioResultBase):
"""Synthesized audio result"""
def to_wav_bytes(self) -> bytes:
"""Convert audio bytes to WAV"""
with io.BytesIO() as wav_io:
wav_file: wave.Wave_write = wave.open(wav_io, "wb")
with wav_file:
@ -76,92 +115,126 @@ class AudioResult(BaseResult, _BaseResultDefaults, _AudioResultBase):
@dataclass
class _MarkResultBase:
"""Result indicating a <mark> has been reached in SSML"""
name: str
"""Name of the <mark>"""
@dataclass
class MarkResult(BaseResult, _BaseResultDefaults, _MarkResultBase):
pass
"""Result indicating a <mark> has been reached in SSML"""
@dataclass
class Voice:
"""Details of a voice in a text to speech system"""
key: str
"""Unique key that can be used to reference the voice"""
name: str
"""Human-readable name of the voice"""
language: str
"""Language of the voice (e.g., en_US)"""
description: str
"""Human-readable description of the voice"""
speakers: typing.Optional[typing.Sequence[str]] = None
"""List of speakers within the voice model if multi-speaker"""
properties: typing.Optional[typing.Mapping[str, typing.Any]] = None
"""Additional properties associated with the voice"""
@property
def is_multispeaker(self) -> bool:
"""True if voice has multiple speakers"""
return (self.speakers is not None) and (len(self.speakers) > 1)
# @dataclass
# class LexiconEntry:
# word: str
# pronunciation: str
# role: typing.Optional[str] = None
# @dataclass
# class Lexicon:
# name: str
# entries: typing.Mapping[str, typing.Sequence[LexiconEntry]]
class TextToSpeechSystem(AbstractContextManager, metaclass=ABCMeta):
"""Abstract base class for open text to speech systems"""
"""Abstract base class for open text to speech systems.
Expected usage:
begin_utterance()
speak_text(...)
add_break(...)
set_mark(...)
speak_tokens(...)
speak_text(...)
results = end_utterance()
In between begin_utterance() and end_utterance(), the voice/language may
also be changed.
"""
@property
@abstractmethod
def voice(self) -> str:
pass
"""Get the current voice key"""
@voice.setter
def voice(self, new_voice: str):
pass
"""Set the current voice key"""
@property
@abstractmethod
def language(self) -> str:
pass
"""Get the current voice language"""
@language.setter
def language(self, new_language: str):
pass
"""Set the current voice language"""
def shutdown(self):
pass
"""Called by the host program when the text to speech system should be stopped"""
def __exit__(self, exc_type, exc_value, traceback):
"""Automatically call shutdown when context manager has exited"""
self.shutdown()
@abstractmethod
def get_voices(self) -> typing.Iterable[Voice]:
pass
"""Returns an iterable of available voices"""
@abstractmethod
def begin_utterance(self):
pass
"""Begins a new utterance"""
@abstractmethod
def speak_text(self, text: str):
pass
"""Speaks text using the underlying system's tokenization mechanism.
Becomes an AudioResult in end_utterance()
"""
@abstractmethod
def speak_tokens(self, tokens: typing.Iterable[BaseToken]):
pass
"""Speak user-defined tokens.
Becomes an AudioResult in end_utterance()
"""
@abstractmethod
def add_break(self, time_ms: int):
pass
"""Add milliseconds of silence to the current utterance.
Becomes an AudioResult in end_utterance()
"""
@abstractmethod
def set_mark(self, name: str):
pass
"""Set a named mark at this point in the utterance.
Becomes a MarkResult in end_utterance()
"""
@abstractmethod
def end_utterance(self) -> typing.Iterable[BaseResult]:
pass
"""Complete an utterance after begin_utterance().
Returns an iterable of results (audio, marks, etc.)
"""