Add more to CLI
This commit is contained in:
parent
bfdc26621e
commit
15a2d22320
8 changed files with 77 additions and 53 deletions
1
mimic3-cli/mimic3_cli/VERSION
Normal file
1
mimic3-cli/mimic3_cli/VERSION
Normal file
|
|
@ -0,0 +1 @@
|
||||||
|
0.1.0
|
||||||
|
|
@ -1 +1,7 @@
|
||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
_DIR = Path(__file__).parent
|
||||||
|
|
||||||
|
__author__ = "Michael Hansen"
|
||||||
|
__version__ = (_DIR / "VERSION").read_text().strip()
|
||||||
|
|
|
||||||
|
|
@ -1,5 +1,6 @@
|
||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
import argparse
|
import argparse
|
||||||
|
import csv
|
||||||
import io
|
import io
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
|
@ -40,6 +41,7 @@ class CommandLineInterfaceState:
|
||||||
texts: typing.Optional[typing.Iterable[str]] = None
|
texts: typing.Optional[typing.Iterable[str]] = None
|
||||||
mark_writer: typing.Optional[typing.TextIO] = None
|
mark_writer: typing.Optional[typing.TextIO] = None
|
||||||
tts: typing.Optional["Mimic3TextToSpeechSystem"] = None
|
tts: typing.Optional["Mimic3TextToSpeechSystem"] = None
|
||||||
|
text_from_stdin: bool = False
|
||||||
|
|
||||||
all_audio: bytes = field(default_factory=bytes)
|
all_audio: bytes = field(default_factory=bytes)
|
||||||
sample_rate_hz: int = 22050
|
sample_rate_hz: int = 22050
|
||||||
|
|
@ -85,39 +87,32 @@ def main():
|
||||||
"""Main entry point"""
|
"""Main entry point"""
|
||||||
args = get_args()
|
args = get_args()
|
||||||
|
|
||||||
# TODO: Print version
|
if args.debug:
|
||||||
|
logging.basicConfig(level=logging.DEBUG)
|
||||||
|
else:
|
||||||
|
logging.basicConfig(level=logging.INFO)
|
||||||
|
|
||||||
# TODO: CUDA support
|
if args.version:
|
||||||
# if args.cuda:
|
# Print version and exit
|
||||||
# import torch
|
from . import __version__
|
||||||
|
|
||||||
# args.cuda = torch.cuda.is_available()
|
print(__version__)
|
||||||
# if not args.cuda:
|
sys.exit(0)
|
||||||
# args.half = False
|
|
||||||
# _LOGGER.warning("CUDA is not available")
|
|
||||||
|
|
||||||
# TODO: Disable Onnx optimizations
|
|
||||||
# Handle optimizations.
|
|
||||||
# onnxruntime crashes on armv7l if optimizations are enabled.
|
|
||||||
# setattr(args, "no_optimizations", False)
|
|
||||||
# if args.optimizations == "off":
|
|
||||||
# args.no_optimizations = True
|
|
||||||
# elif args.optimizations == "auto":
|
|
||||||
# if platform.machine() == "armv7l":
|
|
||||||
# # Enabling optimizations on 32-bit ARM crashes
|
|
||||||
# args.no_optimizations = True
|
|
||||||
|
|
||||||
# TODO: Backend selection
|
|
||||||
# backend: typing.Optional[InferenceBackend] = None
|
|
||||||
# if args.backend:
|
|
||||||
# backend = InferenceBackend(args.backend)
|
|
||||||
|
|
||||||
state = CommandLineInterfaceState(args=args)
|
state = CommandLineInterfaceState(args=args)
|
||||||
initialize_args(state)
|
initialize_args(state)
|
||||||
initialize_tts(state)
|
initialize_tts(state)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
process_lines(state)
|
if args.voices:
|
||||||
|
# Print voices and exit
|
||||||
|
print_voices(state)
|
||||||
|
else:
|
||||||
|
# Process user input
|
||||||
|
if os.isatty(sys.stdin.fileno()):
|
||||||
|
print("Reading text from stdin...", file=sys.stderr)
|
||||||
|
|
||||||
|
process_lines(state)
|
||||||
finally:
|
finally:
|
||||||
shutdown_tts(state)
|
shutdown_tts(state)
|
||||||
|
|
||||||
|
|
@ -158,6 +153,7 @@ def initialize_args(state: CommandLineInterfaceState):
|
||||||
state.texts = args.text
|
state.texts = args.text
|
||||||
else:
|
else:
|
||||||
# Use stdin
|
# Use stdin
|
||||||
|
state.text_from_stdin = True
|
||||||
stdin_format = StdinFormat.LINES
|
stdin_format = StdinFormat.LINES
|
||||||
|
|
||||||
if (args.stdin_format == StdinFormat.AUTO) and args.ssml:
|
if (args.stdin_format == StdinFormat.AUTO) and args.ssml:
|
||||||
|
|
@ -171,9 +167,6 @@ def initialize_args(state: CommandLineInterfaceState):
|
||||||
# Multiple lines
|
# Multiple lines
|
||||||
state.texts = sys.stdin
|
state.texts = sys.stdin
|
||||||
|
|
||||||
if os.isatty(sys.stdin.fileno()):
|
|
||||||
print("Reading text from stdin...", file=sys.stderr)
|
|
||||||
|
|
||||||
assert state.texts is not None
|
assert state.texts is not None
|
||||||
|
|
||||||
if args.process_on_blank_line:
|
if args.process_on_blank_line:
|
||||||
|
|
@ -212,6 +205,10 @@ def initialize_tts(state: CommandLineInterfaceState):
|
||||||
|
|
||||||
state.tts = Mimic3TextToSpeechSystem(Mimic3Settings())
|
state.tts = Mimic3TextToSpeechSystem(Mimic3Settings())
|
||||||
|
|
||||||
|
if args.voices:
|
||||||
|
# Don't bother with the rest of the initialization
|
||||||
|
return
|
||||||
|
|
||||||
if state.args.voice:
|
if state.args.voice:
|
||||||
# Set default voice
|
# Set default voice
|
||||||
state.tts.voice = state.args.voice
|
state.tts.voice = state.args.voice
|
||||||
|
|
@ -414,6 +411,18 @@ def play_wav_bytes(wav_bytes: bytes):
|
||||||
playsound(wav_file.name)
|
playsound(wav_file.name)
|
||||||
|
|
||||||
|
|
||||||
|
def print_voices(state: CommandLineInterfaceState):
|
||||||
|
assert state.tts is not None
|
||||||
|
|
||||||
|
voices = list(state.tts.get_voices())
|
||||||
|
voices = sorted(voices, key=lambda v: v.key)
|
||||||
|
|
||||||
|
writer = csv.writer(sys.stdout, delimiter="\t")
|
||||||
|
writer.writerow(("KEY", "LANGUAGE", "NAME", "DESCRIPTION"))
|
||||||
|
for voice in voices:
|
||||||
|
writer.writerow((voice.key, voice.language, voice.name, voice.description))
|
||||||
|
|
||||||
|
|
||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -441,9 +450,7 @@ def get_args():
|
||||||
# "--voices-dir",
|
# "--voices-dir",
|
||||||
# help="Directory with voices (format is <language>/<name_model-type>)",
|
# help="Directory with voices (format is <language>/<name_model-type>)",
|
||||||
# )
|
# )
|
||||||
# parser.add_argument(
|
parser.add_argument("--voices", action="store_true", help="List available voices")
|
||||||
# "--list", action="store_true", help="List available voices/vocoders"
|
|
||||||
# )
|
|
||||||
parser.add_argument("--output-dir", help="Directory to write WAV file(s)")
|
parser.add_argument("--output-dir", help="Directory to write WAV file(s)")
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--output-naming",
|
"--output-naming",
|
||||||
|
|
@ -511,29 +518,12 @@ def get_args():
|
||||||
"--preload-voice", action="append", help="Preload voice when starting up"
|
"--preload-voice", action="append", help="Preload voice when starting up"
|
||||||
)
|
)
|
||||||
parser.add_argument("--seed", type=int, help="Set random seed (default: not set)")
|
parser.add_argument("--seed", type=int, help="Set random seed (default: not set)")
|
||||||
# parser.add_argument("--version", action="store_true", help="Print version and exit")
|
parser.add_argument("--version", action="store_true", help="Print version and exit")
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--debug", action="store_true", help="Print DEBUG messages to the console"
|
"--debug", action="store_true", help="Print DEBUG messages to the console"
|
||||||
)
|
)
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
if args.debug:
|
return parser.parse_args()
|
||||||
logging.basicConfig(level=logging.DEBUG)
|
|
||||||
else:
|
|
||||||
logging.basicConfig(level=logging.INFO)
|
|
||||||
|
|
||||||
# -------------------------------------------------------------------------
|
|
||||||
|
|
||||||
# if args.version:
|
|
||||||
# # Print version and exit
|
|
||||||
# from larynx import __version__
|
|
||||||
|
|
||||||
# print(__version__)
|
|
||||||
# sys.exit(0)
|
|
||||||
|
|
||||||
# -------------------------------------------------------------------------
|
|
||||||
|
|
||||||
return args
|
|
||||||
|
|
||||||
|
|
||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
|
|
|
||||||
1
mimic3-http/mimic3_http/VERSION
Normal file
1
mimic3-http/mimic3_http/VERSION
Normal file
|
|
@ -0,0 +1 @@
|
||||||
|
0.1.0
|
||||||
|
|
@ -138,9 +138,6 @@ _WAV_CACHE: typing.Dict[TextToWavParams, Path] = {}
|
||||||
|
|
||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
|
|
||||||
# _TTS: typing.Dict[str, Mimic3] = {}
|
|
||||||
# _VOICE: str = args.voice
|
|
||||||
|
|
||||||
|
|
||||||
# TODO: XDG voice directories
|
# TODO: XDG voice directories
|
||||||
# TODO: args.voices_dir
|
# TODO: args.voices_dir
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,2 @@
|
||||||
|
GenericExecuteSynth "printf %s \'$DATA\' | /home/hansenm/opt/mimic3/mimic3-http/client.sh --voice \'$VOICE\' --length-scale 0.5 --stdout | $PLAY_COMMAND"
|
||||||
|
AddVoice "en-us" "FEMALE1" "en_US/amy_low"
|
||||||
|
|
@ -0,0 +1,2 @@
|
||||||
|
DefaultVoiceType "FEMALE1"
|
||||||
|
DefaultModule mimic3-generic
|
||||||
25
mimic3-tts/README.md
Normal file
25
mimic3-tts/README.md
Normal file
|
|
@ -0,0 +1,25 @@
|
||||||
|
# Mimic 3
|
||||||
|
|
||||||
|
A fast and local neural text to speech system for Mycroft.
|
||||||
|
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
|
||||||
|
Mimic 3 uses the [VITS](https://arxiv.org/abs/2106.06103), a "Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech". VITS is a combination of the [GlowTTS duration predictor](https://arxiv.org/abs/2005.11129) and the [HiFi-GAN vocoder](https://arxiv.org/abs/2010.05646).
|
||||||
|
|
||||||
|
Our implementation is heavily based on [Jaehyeon Kim's PyTorch model](https://github.com/jaywalnut310/vits), with the addition of [Onnx runtime](https://onnxruntime.ai/) export for speed.
|
||||||
|
|
||||||
|
|
||||||
|
### gruut Phoneme-based Voices
|
||||||
|
|
||||||
|
Voices that use [gruut](https://github.com/rhasspy/gruut/) for phonemization.
|
||||||
|
|
||||||
|
|
||||||
|
### eSpeak Phoneme-based Voices
|
||||||
|
|
||||||
|
Voices that use [eSpeak-ng](https://github.com/espeak-ng/espeak-ng) for phonemization (via [espeak-phonemizer](https://github.com/rhasspy/espeak-phonemizer)).
|
||||||
|
|
||||||
|
|
||||||
|
### Character-based Voices
|
||||||
|
|
||||||
|
Voices whose "phonemes" are characters from an alphabet, typically with some punctuation.
|
||||||
Loading…
Add table
Add a link
Reference in a new issue