diff --git a/mimic3-cli/mimic3_cli/VERSION b/mimic3-cli/mimic3_cli/VERSION new file mode 100644 index 0000000..6e8bf73 --- /dev/null +++ b/mimic3-cli/mimic3_cli/VERSION @@ -0,0 +1 @@ +0.1.0 diff --git a/mimic3-cli/mimic3_cli/__init__.py b/mimic3-cli/mimic3_cli/__init__.py index e5a0d9b..6044114 100644 --- a/mimic3-cli/mimic3_cli/__init__.py +++ b/mimic3-cli/mimic3_cli/__init__.py @@ -1 +1,7 @@ #!/usr/bin/env python3 +from pathlib import Path + +_DIR = Path(__file__).parent + +__author__ = "Michael Hansen" +__version__ = (_DIR / "VERSION").read_text().strip() diff --git a/mimic3-cli/mimic3_cli/__main__.py b/mimic3-cli/mimic3_cli/__main__.py index 7614851..01c7fb0 100644 --- a/mimic3-cli/mimic3_cli/__main__.py +++ b/mimic3-cli/mimic3_cli/__main__.py @@ -1,5 +1,6 @@ #!/usr/bin/env python3 import argparse +import csv import io import logging import os @@ -40,6 +41,7 @@ class CommandLineInterfaceState: texts: typing.Optional[typing.Iterable[str]] = None mark_writer: typing.Optional[typing.TextIO] = None tts: typing.Optional["Mimic3TextToSpeechSystem"] = None + text_from_stdin: bool = False all_audio: bytes = field(default_factory=bytes) sample_rate_hz: int = 22050 @@ -85,39 +87,32 @@ def main(): """Main entry point""" args = get_args() - # TODO: Print version + if args.debug: + logging.basicConfig(level=logging.DEBUG) + else: + logging.basicConfig(level=logging.INFO) - # TODO: CUDA support - # if args.cuda: - # import torch + if args.version: + # Print version and exit + from . import __version__ - # args.cuda = torch.cuda.is_available() - # if not args.cuda: - # args.half = False - # _LOGGER.warning("CUDA is not available") - - # TODO: Disable Onnx optimizations - # Handle optimizations. - # onnxruntime crashes on armv7l if optimizations are enabled. - # setattr(args, "no_optimizations", False) - # if args.optimizations == "off": - # args.no_optimizations = True - # elif args.optimizations == "auto": - # if platform.machine() == "armv7l": - # # Enabling optimizations on 32-bit ARM crashes - # args.no_optimizations = True - - # TODO: Backend selection - # backend: typing.Optional[InferenceBackend] = None - # if args.backend: - # backend = InferenceBackend(args.backend) + print(__version__) + sys.exit(0) state = CommandLineInterfaceState(args=args) initialize_args(state) initialize_tts(state) try: - process_lines(state) + if args.voices: + # Print voices and exit + print_voices(state) + else: + # Process user input + if os.isatty(sys.stdin.fileno()): + print("Reading text from stdin...", file=sys.stderr) + + process_lines(state) finally: shutdown_tts(state) @@ -158,6 +153,7 @@ def initialize_args(state: CommandLineInterfaceState): state.texts = args.text else: # Use stdin + state.text_from_stdin = True stdin_format = StdinFormat.LINES if (args.stdin_format == StdinFormat.AUTO) and args.ssml: @@ -171,9 +167,6 @@ def initialize_args(state: CommandLineInterfaceState): # Multiple lines state.texts = sys.stdin - if os.isatty(sys.stdin.fileno()): - print("Reading text from stdin...", file=sys.stderr) - assert state.texts is not None if args.process_on_blank_line: @@ -212,6 +205,10 @@ def initialize_tts(state: CommandLineInterfaceState): state.tts = Mimic3TextToSpeechSystem(Mimic3Settings()) + if args.voices: + # Don't bother with the rest of the initialization + return + if state.args.voice: # Set default voice state.tts.voice = state.args.voice @@ -414,6 +411,18 @@ def play_wav_bytes(wav_bytes: bytes): playsound(wav_file.name) +def print_voices(state: CommandLineInterfaceState): + assert state.tts is not None + + voices = list(state.tts.get_voices()) + voices = sorted(voices, key=lambda v: v.key) + + writer = csv.writer(sys.stdout, delimiter="\t") + writer.writerow(("KEY", "LANGUAGE", "NAME", "DESCRIPTION")) + for voice in voices: + writer.writerow((voice.key, voice.language, voice.name, voice.description)) + + # ----------------------------------------------------------------------------- @@ -441,9 +450,7 @@ def get_args(): # "--voices-dir", # help="Directory with voices (format is /)", # ) - # parser.add_argument( - # "--list", action="store_true", help="List available voices/vocoders" - # ) + parser.add_argument("--voices", action="store_true", help="List available voices") parser.add_argument("--output-dir", help="Directory to write WAV file(s)") parser.add_argument( "--output-naming", @@ -511,29 +518,12 @@ def get_args(): "--preload-voice", action="append", help="Preload voice when starting up" ) parser.add_argument("--seed", type=int, help="Set random seed (default: not set)") - # parser.add_argument("--version", action="store_true", help="Print version and exit") + parser.add_argument("--version", action="store_true", help="Print version and exit") parser.add_argument( "--debug", action="store_true", help="Print DEBUG messages to the console" ) - args = parser.parse_args() - if args.debug: - logging.basicConfig(level=logging.DEBUG) - else: - logging.basicConfig(level=logging.INFO) - - # ------------------------------------------------------------------------- - - # if args.version: - # # Print version and exit - # from larynx import __version__ - - # print(__version__) - # sys.exit(0) - - # ------------------------------------------------------------------------- - - return args + return parser.parse_args() # ----------------------------------------------------------------------------- diff --git a/mimic3-http/mimic3_http/VERSION b/mimic3-http/mimic3_http/VERSION new file mode 100644 index 0000000..6e8bf73 --- /dev/null +++ b/mimic3-http/mimic3_http/VERSION @@ -0,0 +1 @@ +0.1.0 diff --git a/mimic3-http/mimic3_http/__main__.py b/mimic3-http/mimic3_http/__main__.py index af632dc..0ea0958 100644 --- a/mimic3-http/mimic3_http/__main__.py +++ b/mimic3-http/mimic3_http/__main__.py @@ -138,9 +138,6 @@ _WAV_CACHE: typing.Dict[TextToWavParams, Path] = {} # ----------------------------------------------------------------------------- -# _TTS: typing.Dict[str, Mimic3] = {} -# _VOICE: str = args.voice - # TODO: XDG voice directories # TODO: args.voices_dir diff --git a/mimic3-http/mimic3_http/examples/speech-dispatcher/mimic3-generic.conf b/mimic3-http/mimic3_http/examples/speech-dispatcher/mimic3-generic.conf new file mode 100644 index 0000000..ab8a662 --- /dev/null +++ b/mimic3-http/mimic3_http/examples/speech-dispatcher/mimic3-generic.conf @@ -0,0 +1,2 @@ +GenericExecuteSynth "printf %s \'$DATA\' | /home/hansenm/opt/mimic3/mimic3-http/client.sh --voice \'$VOICE\' --length-scale 0.5 --stdout | $PLAY_COMMAND" +AddVoice "en-us" "FEMALE1" "en_US/amy_low" diff --git a/mimic3-http/mimic3_http/examples/speech-dispatcher/speechd.conf b/mimic3-http/mimic3_http/examples/speech-dispatcher/speechd.conf new file mode 100644 index 0000000..fe1bf6a --- /dev/null +++ b/mimic3-http/mimic3_http/examples/speech-dispatcher/speechd.conf @@ -0,0 +1,2 @@ +DefaultVoiceType "FEMALE1" +DefaultModule mimic3-generic diff --git a/mimic3-tts/README.md b/mimic3-tts/README.md new file mode 100644 index 0000000..8606767 --- /dev/null +++ b/mimic3-tts/README.md @@ -0,0 +1,25 @@ +# Mimic 3 + +A fast and local neural text to speech system for Mycroft. + + +## Architecture + +Mimic 3 uses the [VITS](https://arxiv.org/abs/2106.06103), a "Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech". VITS is a combination of the [GlowTTS duration predictor](https://arxiv.org/abs/2005.11129) and the [HiFi-GAN vocoder](https://arxiv.org/abs/2010.05646). + +Our implementation is heavily based on [Jaehyeon Kim's PyTorch model](https://github.com/jaywalnut310/vits), with the addition of [Onnx runtime](https://onnxruntime.ai/) export for speed. + + +### gruut Phoneme-based Voices + +Voices that use [gruut](https://github.com/rhasspy/gruut/) for phonemization. + + +### eSpeak Phoneme-based Voices + +Voices that use [eSpeak-ng](https://github.com/espeak-ng/espeak-ng) for phonemization (via [espeak-phonemizer](https://github.com/rhasspy/espeak-phonemizer)). + + +### Character-based Voices + +Voices whose "phonemes" are characters from an alphabet, typically with some punctuation.