diff --git a/mimic3_http/app.py b/mimic3_http/app.py index 8ca1c73..83d7ecd 100644 --- a/mimic3_http/app.py +++ b/mimic3_http/app.py @@ -19,6 +19,8 @@ import dataclasses import json import logging import re +import shlex +import subprocess import typing from pathlib import Path from queue import Queue @@ -163,7 +165,7 @@ def get_app(args: argparse.Namespace, request_queue: Queue, temp_dir: str): ) @app.route("/api/tts", methods=["GET", "POST"]) - async def app_tts() -> Response: + async def app_tts() -> typing.Union[Response, str]: """Speak text to WAV.""" tts_args: typing.Dict[str, typing.Any] = { "length_scale": args.length_scale, @@ -224,7 +226,15 @@ def get_app(args: argparse.Namespace, request_queue: Queue, temp_dir: str): TextToWavParams(text=text, **tts_args), no_cache=no_cache ) - return Response(wav_bytes, mimetype="audio/wav") + audio_target = request.args.get("audioTarget", "client").strip().lower() + if audio_target == "client": + return Response(wav_bytes, mimetype="audio/wav") + + # Play audio on server + play_cmd = shlex.split(args.play_program) + subprocess.run(play_cmd, input=wav_bytes, check=True) + + return "OK" @app.route("/api/voices", methods=["GET"]) async def api_voices(): diff --git a/mimic3_http/args.py b/mimic3_http/args.py index f854024..d47f531 100644 --- a/mimic3_http/args.py +++ b/mimic3_http/args.py @@ -90,6 +90,9 @@ def get_args(argv=None) -> argparse.Namespace: "--default-voice", help="Default voice key to select in web interface", ) + parser.add_argument( + "--play-program", default="aplay -q", help="Program to play WAV audio on server" + ) parser.add_argument( "--no-show-openapi", action="store_true", help="Don't show OpenAPI link" ) diff --git a/mimic3_http/templates/index.html b/mimic3_http/templates/index.html index 9b5e26a..bf2eb45 100644 --- a/mimic3_http/templates/index.html +++ b/mimic3_http/templates/index.html @@ -115,6 +115,15 @@ +