open source
This commit is contained in:
parent
70b6e17c69
commit
a93bfc1ec9
61 changed files with 4013 additions and 126 deletions
250
vocode/streaming/synthesizer/azure_synthesizer.py
Normal file
250
vocode/streaming/synthesizer/azure_synthesizer.py
Normal file
|
|
@ -0,0 +1,250 @@
|
|||
import logging
|
||||
import os
|
||||
import re
|
||||
from typing import Any, Optional
|
||||
from xml.etree import ElementTree
|
||||
import azure.cognitiveservices.speech as speechsdk
|
||||
from dotenv import load_dotenv
|
||||
|
||||
from vocode.streaming.agent.bot_sentiment_analyser import BotSentiment
|
||||
from vocode.streaming.models.message import BaseMessage, SSMLMessage
|
||||
|
||||
from vocode.streaming.synthesizer.base_synthesizer import (
|
||||
BaseSynthesizer,
|
||||
SynthesisResult,
|
||||
FILLER_PHRASES,
|
||||
FILLER_AUDIO_PATH,
|
||||
FillerAudio,
|
||||
encode_as_wav,
|
||||
)
|
||||
from vocode.streaming.models.synthesizer import AzureSynthesizerConfig
|
||||
from vocode.streaming.models.audio_encoding import AudioEncoding
|
||||
|
||||
load_dotenv()
|
||||
|
||||
NAMESPACES = {
|
||||
"mstts": "https://www.w3.org/2001/mstts",
|
||||
"": "https://www.w3.org/2001/10/synthesis",
|
||||
}
|
||||
|
||||
ElementTree.register_namespace("", NAMESPACES.get(""))
|
||||
ElementTree.register_namespace("mstts", NAMESPACES.get("mstts"))
|
||||
|
||||
|
||||
class WordBoundaryEventPool:
|
||||
def __init__(self):
|
||||
self.events = []
|
||||
|
||||
def add(self, event):
|
||||
self.events.append(
|
||||
{
|
||||
"text": event.text,
|
||||
"text_offset": event.text_offset,
|
||||
"audio_offset": (event.audio_offset + 5000) / (10000 * 1000),
|
||||
"boudary_type": event.boundary_type,
|
||||
}
|
||||
)
|
||||
|
||||
def get_events_sorted(self):
|
||||
return sorted(self.events, key=lambda event: event["audio_offset"])
|
||||
|
||||
|
||||
class AzureSynthesizer(BaseSynthesizer):
|
||||
OFFSET_MS = 100
|
||||
|
||||
def __init__(
|
||||
self, synthesizer_config: AzureSynthesizerConfig, logger: logging.Logger = None
|
||||
):
|
||||
super().__init__(synthesizer_config)
|
||||
self.synthesizer_config = synthesizer_config
|
||||
# Instantiates a client
|
||||
speech_config = speechsdk.SpeechConfig(
|
||||
subscription=os.environ.get("AZURE_SPEECH_KEY"),
|
||||
region=os.environ.get("AZURE_SPEECH_REGION"),
|
||||
)
|
||||
if self.synthesizer_config.audio_encoding == AudioEncoding.LINEAR16:
|
||||
if self.synthesizer_config.sampling_rate == 44100:
|
||||
speech_config.set_speech_synthesis_output_format(
|
||||
speechsdk.SpeechSynthesisOutputFormat.Raw44100Hz16BitMonoPcm
|
||||
)
|
||||
if self.synthesizer_config.sampling_rate == 48000:
|
||||
speech_config.set_speech_synthesis_output_format(
|
||||
speechsdk.SpeechSynthesisOutputFormat.Raw48Khz16BitMonoPcm
|
||||
)
|
||||
if self.synthesizer_config.sampling_rate == 24000:
|
||||
speech_config.set_speech_synthesis_output_format(
|
||||
speechsdk.SpeechSynthesisOutputFormat.Raw24Khz16BitMonoPcm
|
||||
)
|
||||
elif self.synthesizer_config.sampling_rate == 16000:
|
||||
speech_config.set_speech_synthesis_output_format(
|
||||
speechsdk.SpeechSynthesisOutputFormat.Raw16Khz16BitMonoPcm
|
||||
)
|
||||
elif self.synthesizer_config.sampling_rate == 8000:
|
||||
speech_config.set_speech_synthesis_output_format(
|
||||
speechsdk.SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm
|
||||
)
|
||||
elif self.synthesizer_config.audio_encoding == AudioEncoding.MULAW:
|
||||
speech_config.set_speech_synthesis_output_format(
|
||||
speechsdk.SpeechSynthesisOutputFormat.Raw8Khz8BitMonoMULaw
|
||||
)
|
||||
self.synthesizer = speechsdk.SpeechSynthesizer(
|
||||
speech_config=speech_config, audio_config=None
|
||||
)
|
||||
|
||||
self.voice_name = self.synthesizer_config.voice_name
|
||||
self.pitch = self.synthesizer_config.pitch
|
||||
self.rate = self.synthesizer_config.rate
|
||||
self.logger = logger or logging.getLogger(__name__)
|
||||
|
||||
def get_phrase_filler_audios(self) -> list[FillerAudio]:
|
||||
filler_phrase_audios = []
|
||||
for filler_phrase in FILLER_PHRASES:
|
||||
cache_key = "-".join(
|
||||
(
|
||||
str(filler_phrase.text),
|
||||
str(self.synthesizer_config.type),
|
||||
str(self.synthesizer_config.audio_encoding),
|
||||
str(self.synthesizer_config.sampling_rate),
|
||||
str(self.voice_name),
|
||||
str(self.pitch),
|
||||
str(self.rate),
|
||||
)
|
||||
)
|
||||
filler_audio_path = os.path.join(FILLER_AUDIO_PATH, f"{cache_key}.bytes")
|
||||
if os.path.exists(filler_audio_path):
|
||||
audio_data = open(filler_audio_path, "rb").read()
|
||||
else:
|
||||
self.logger.debug(f"Generating filler audio for {filler_phrase.text}")
|
||||
ssml = self.create_ssml(filler_phrase.text)
|
||||
result = self.synthesizer.speak_ssml(ssml)
|
||||
offset = self.synthesizer_config.sampling_rate * self.OFFSET_MS // 1000
|
||||
audio_data = result.audio_data[offset:]
|
||||
with open(filler_audio_path, "wb") as f:
|
||||
f.write(audio_data)
|
||||
filler_phrase_audios.append(
|
||||
FillerAudio(
|
||||
filler_phrase,
|
||||
audio_data,
|
||||
self.synthesizer_config,
|
||||
)
|
||||
)
|
||||
return filler_phrase_audios
|
||||
|
||||
def add_marks(self, message: str, index=0) -> str:
|
||||
search_result = re.search(r"([\.\,\:\;\-\—]+)", message)
|
||||
if search_result is None:
|
||||
return message
|
||||
start, end = search_result.span()
|
||||
with_mark = message[:start] + f'<mark name="{index}" />' + message[start:end]
|
||||
rest = message[end:]
|
||||
rest_stripped = re.sub(r"^(.+)([\.\,\:\;\-\—]+)$", r"\1", rest)
|
||||
if len(rest_stripped) == 0:
|
||||
return with_mark
|
||||
return with_mark + self.add_marks(rest_stripped, index + 1)
|
||||
|
||||
def word_boundary_cb(self, evt, pool):
|
||||
pool.add(evt)
|
||||
|
||||
def create_ssml(
|
||||
self, message: str, bot_sentiment: Optional[BotSentiment] = None
|
||||
) -> str:
|
||||
ssml_root = ElementTree.fromstring(
|
||||
'<speak version="1.0" xmlns="https://www.w3.org/2001/10/synthesis" xml:lang="en-US"></speak>'
|
||||
)
|
||||
voice = ElementTree.SubElement(ssml_root, "voice")
|
||||
voice.set("name", self.voice_name)
|
||||
voice_root = voice
|
||||
if bot_sentiment and bot_sentiment.emotion:
|
||||
styled = ElementTree.SubElement(
|
||||
voice, "{%s}express-as" % NAMESPACES.get("mstts")
|
||||
)
|
||||
styled.set("style", bot_sentiment.emotion)
|
||||
styled.set(
|
||||
"styledegree", str(bot_sentiment.degree * 2)
|
||||
) # Azure specific, it's a scale of 0-2
|
||||
voice_root = styled
|
||||
prosody = ElementTree.SubElement(voice_root, "prosody")
|
||||
prosody.set("pitch", f"{self.pitch}%")
|
||||
prosody.set("rate", f"{self.rate}%")
|
||||
prosody.text = message.strip()
|
||||
return ElementTree.tostring(ssml_root, encoding="unicode")
|
||||
|
||||
def synthesize_ssml(self, ssml: str) -> tuple[speechsdk.AudioDataStream, str]:
|
||||
result = self.synthesizer.start_speaking_ssml_async(ssml).get()
|
||||
return speechsdk.AudioDataStream(result)
|
||||
|
||||
def ready_synthesizer(self):
|
||||
connection = speechsdk.Connection.from_speech_synthesizer(self.synthesizer)
|
||||
connection.open(True)
|
||||
|
||||
# given the number of seconds the message was allowed to go until, where did we get in the message?
|
||||
def get_message_up_to(
|
||||
self,
|
||||
message: str,
|
||||
ssml: str,
|
||||
seconds: int,
|
||||
word_boundary_event_pool: WordBoundaryEventPool,
|
||||
) -> str:
|
||||
events = word_boundary_event_pool.get_events_sorted()
|
||||
for event in events:
|
||||
if event["audio_offset"] > seconds:
|
||||
ssml_fragment = ssml[: event["text_offset"]]
|
||||
return ssml_fragment.split(">")[-1]
|
||||
return message
|
||||
|
||||
def create_speech(
|
||||
self,
|
||||
message: BaseMessage,
|
||||
chunk_size: int,
|
||||
bot_sentiment: Optional[BotSentiment] = None,
|
||||
) -> SynthesisResult:
|
||||
# offset = int(self.OFFSET_MS * (self.synthesizer_config.sampling_rate / 1000))
|
||||
offset = 0
|
||||
self.logger.debug(f"Synthesizing message: {message}")
|
||||
|
||||
def chunk_generator(
|
||||
audio_data_stream: speechsdk.AudioDataStream, chunk_transform=lambda x: x
|
||||
):
|
||||
audio_buffer = bytes(chunk_size)
|
||||
filled_size = audio_data_stream.read_data(audio_buffer)
|
||||
if filled_size != chunk_size:
|
||||
yield SynthesisResult.ChunkResult(
|
||||
chunk_transform(audio_buffer[offset:]), True
|
||||
)
|
||||
return
|
||||
else:
|
||||
yield SynthesisResult.ChunkResult(
|
||||
chunk_transform(audio_buffer[offset:]), False
|
||||
)
|
||||
while True:
|
||||
filled_size = audio_data_stream.read_data(audio_buffer)
|
||||
if filled_size != chunk_size:
|
||||
yield SynthesisResult.ChunkResult(
|
||||
chunk_transform(audio_buffer[: filled_size - offset]), True
|
||||
)
|
||||
break
|
||||
yield SynthesisResult.ChunkResult(chunk_transform(audio_buffer), False)
|
||||
|
||||
word_boundary_event_pool = WordBoundaryEventPool()
|
||||
self.synthesizer.synthesis_word_boundary.connect(
|
||||
lambda event: self.word_boundary_cb(event, word_boundary_event_pool)
|
||||
)
|
||||
ssml = (
|
||||
message.ssml
|
||||
if isinstance(message, SSMLMessage)
|
||||
else self.create_ssml(message.text, bot_sentiment=bot_sentiment)
|
||||
)
|
||||
audio_data_stream = self.synthesize_ssml(ssml)
|
||||
if self.synthesizer_config.should_encode_as_wav:
|
||||
output_generator = chunk_generator(
|
||||
audio_data_stream,
|
||||
lambda chunk: encode_as_wav(chunk, self.synthesizer_config),
|
||||
)
|
||||
else:
|
||||
output_generator = chunk_generator(audio_data_stream)
|
||||
return SynthesisResult(
|
||||
output_generator,
|
||||
lambda seconds: self.get_message_up_to(
|
||||
message, ssml, seconds, word_boundary_event_pool
|
||||
),
|
||||
)
|
||||
169
vocode/streaming/synthesizer/base_synthesizer.py
Normal file
169
vocode/streaming/synthesizer/base_synthesizer.py
Normal file
|
|
@ -0,0 +1,169 @@
|
|||
import os
|
||||
from typing import Any, Generator, Callable, Optional
|
||||
import math
|
||||
import io
|
||||
import wave
|
||||
from nltk.tokenize import word_tokenize
|
||||
from nltk.tokenize.treebank import TreebankWordDetokenizer
|
||||
|
||||
from vocode.streaming.agent.bot_sentiment_analyser import BotSentiment
|
||||
from vocode.streaming.models.agent import FillerAudioConfig
|
||||
from vocode.streaming.models.message import BaseMessage
|
||||
from vocode.streaming.utils import convert_wav, get_chunk_size_per_second
|
||||
from vocode.streaming.models.audio_encoding import AudioEncoding
|
||||
from vocode.streaming.models.synthesizer import SynthesizerConfig
|
||||
|
||||
FILLER_PHRASES = [
|
||||
BaseMessage(text="Um..."),
|
||||
BaseMessage(text="Uh..."),
|
||||
BaseMessage(text="Uh-huh..."),
|
||||
BaseMessage(text="Mm-hmm..."),
|
||||
BaseMessage(text="Hmm..."),
|
||||
BaseMessage(text="Okay..."),
|
||||
BaseMessage(text="Right..."),
|
||||
BaseMessage(text="Let me see..."),
|
||||
]
|
||||
FILLER_AUDIO_PATH = os.path.join(os.path.dirname(__file__), "filler_audio")
|
||||
TYPING_NOISE_PATH = "%s/typing-noise.wav" % FILLER_AUDIO_PATH
|
||||
|
||||
|
||||
def encode_as_wav(chunk: bytes, synthesizer_config: SynthesizerConfig) -> bytes:
|
||||
output_bytes_io = io.BytesIO()
|
||||
in_memory_wav = wave.open(output_bytes_io, "wb")
|
||||
in_memory_wav.setnchannels(1)
|
||||
assert synthesizer_config.audio_encoding == AudioEncoding.LINEAR16
|
||||
in_memory_wav.setsampwidth(2)
|
||||
in_memory_wav.setframerate(synthesizer_config.sampling_rate)
|
||||
in_memory_wav.writeframes(chunk)
|
||||
output_bytes_io.seek(0)
|
||||
return output_bytes_io.read()
|
||||
|
||||
|
||||
class SynthesisResult:
|
||||
class ChunkResult:
|
||||
def __init__(self, chunk: bytes, is_last_chunk: bool):
|
||||
self.chunk = chunk
|
||||
self.is_last_chunk = is_last_chunk
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
chunk_generator: Generator[ChunkResult, None, None],
|
||||
get_message_up_to: Callable[[int], str],
|
||||
):
|
||||
self.chunk_generator = chunk_generator
|
||||
self.get_message_up_to = get_message_up_to
|
||||
|
||||
|
||||
class FillerAudio:
|
||||
def __init__(
|
||||
self,
|
||||
message: BaseMessage,
|
||||
audio_data: bytes,
|
||||
synthesizer_config: SynthesizerConfig,
|
||||
is_interruptable: bool = False,
|
||||
seconds_per_chunk: int = 1,
|
||||
):
|
||||
self.message = message
|
||||
self.audio_data = audio_data
|
||||
self.synthesizer_config = synthesizer_config
|
||||
self.is_interruptable = is_interruptable
|
||||
self.seconds_per_chunk = seconds_per_chunk
|
||||
|
||||
def create_synthesis_result(self) -> SynthesisResult:
|
||||
chunk_size = (
|
||||
get_chunk_size_per_second(
|
||||
self.synthesizer_config.audio_encoding,
|
||||
self.synthesizer_config.sampling_rate,
|
||||
)
|
||||
* self.seconds_per_chunk
|
||||
)
|
||||
|
||||
def chunk_generator(chunk_transform=lambda x: x):
|
||||
for i in range(0, len(self.audio_data), chunk_size):
|
||||
if i + chunk_size > len(self.audio_data):
|
||||
yield SynthesisResult.ChunkResult(
|
||||
chunk_transform(self.audio_data[i:]), True
|
||||
)
|
||||
else:
|
||||
yield SynthesisResult.ChunkResult(
|
||||
chunk_transform(self.audio_data[i : i + chunk_size]), False
|
||||
)
|
||||
|
||||
if self.synthesizer_config.should_encode_as_wav:
|
||||
output_generator = chunk_generator(
|
||||
lambda chunk: encode_as_wav(chunk, self.synthesizer_config)
|
||||
)
|
||||
else:
|
||||
output_generator = chunk_generator()
|
||||
return SynthesisResult(output_generator, lambda seconds: self.message.text)
|
||||
|
||||
|
||||
class BaseSynthesizer:
|
||||
def __init__(self, synthesizer_config: SynthesizerConfig):
|
||||
self.synthesizer_config = synthesizer_config
|
||||
if synthesizer_config.audio_encoding == AudioEncoding.MULAW:
|
||||
assert (
|
||||
synthesizer_config.sampling_rate == 8000
|
||||
), "MuLaw encoding only supports 8kHz sampling rate"
|
||||
self.filler_audios: list[FillerAudio] = []
|
||||
|
||||
def get_synthesizer_config(self) -> SynthesizerConfig:
|
||||
return self.synthesizer_config
|
||||
|
||||
def get_typing_noise_filler_audio(self) -> FillerAudio:
|
||||
return FillerAudio(
|
||||
message=BaseMessage(text="<typing noise>"),
|
||||
audio_data=convert_wav(
|
||||
TYPING_NOISE_PATH,
|
||||
output_sample_rate=self.synthesizer_config.sampling_rate,
|
||||
output_encoding=self.synthesizer_config.audio_encoding,
|
||||
),
|
||||
synthesizer_config=self.synthesizer_config,
|
||||
is_interruptable=True,
|
||||
seconds_per_chunk=2,
|
||||
)
|
||||
|
||||
def set_filler_audios(self, filler_audio_config: FillerAudioConfig):
|
||||
if filler_audio_config.use_phrases:
|
||||
self.filler_audios = self.get_phrase_filler_audios()
|
||||
elif filler_audio_config.use_typing_noise:
|
||||
self.filler_audios = [self.get_typing_noise_filler_audio()]
|
||||
|
||||
def get_phrase_filler_audios(self) -> list[FillerAudio]:
|
||||
return []
|
||||
|
||||
def ready_synthesizer(self):
|
||||
pass
|
||||
|
||||
# given the number of seconds the message was allowed to go until, where did we get in the message?
|
||||
def get_message_cutoff_from_total_response_length(
|
||||
self, message: BaseMessage, seconds: int, size_of_output: int
|
||||
) -> str:
|
||||
estimated_output_seconds = (
|
||||
size_of_output / self.synthesizer_config.sampling_rate
|
||||
)
|
||||
estimated_output_seconds_per_char = estimated_output_seconds / len(message.text)
|
||||
return message.text[: int(seconds / estimated_output_seconds_per_char)]
|
||||
|
||||
def get_message_cutoff_from_voice_speed(
|
||||
self, message: BaseMessage, seconds: int, words_per_minute: int
|
||||
) -> str:
|
||||
words_per_second = words_per_minute / 60
|
||||
estimated_words_spoken = math.floor(words_per_second * seconds)
|
||||
tokens = word_tokenize(message.text)
|
||||
return TreebankWordDetokenizer().detokenize(tokens[:estimated_words_spoken])
|
||||
|
||||
def get_maybe_cached_synthesis_result(
|
||||
self, message: BaseMessage, chunk_size: int
|
||||
) -> Optional[SynthesisResult]:
|
||||
return
|
||||
|
||||
# returns a chunk generator and a thunk that can tell you what part of the message was read given the number of seconds spoken
|
||||
# chunk generator must return tuple (bytes of size chunk_size, flag if it is the last chunk)
|
||||
def create_speech(
|
||||
self,
|
||||
message: BaseMessage,
|
||||
chunk_size: int,
|
||||
bot_sentiment: Optional[BotSentiment] = None,
|
||||
) -> SynthesisResult:
|
||||
raise NotImplementedError
|
||||
50
vocode/streaming/synthesizer/eleven_labs_synthesizer.py
Normal file
50
vocode/streaming/synthesizer/eleven_labs_synthesizer.py
Normal file
|
|
@ -0,0 +1,50 @@
|
|||
from typing import Any, Optional
|
||||
import os
|
||||
from dotenv import load_dotenv
|
||||
import requests
|
||||
|
||||
from vocode.streaming.synthesizer.base_synthesizer import (
|
||||
BaseSynthesizer,
|
||||
SynthesisResult,
|
||||
)
|
||||
from vocode.streaming.models.synthesizer import ElevenLabsSynthesizerConfig
|
||||
from vocode.streaming.agent.bot_sentiment_analyser import BotSentiment
|
||||
from vocode.streaming.models.message import BaseMessage
|
||||
|
||||
load_dotenv()
|
||||
|
||||
ELEVEN_LABS_API_KEY = os.environ.get("ELEVEN_LABS_API_KEY")
|
||||
ELEVEN_LABS_BASE_URL = "https://api.elevenlabs.io/v1/"
|
||||
ADAM_VOICE_ID = "pNInz6obpgDQGcFmaJgB"
|
||||
OBAMA_VOICE_ID = "vLITIS0SH2an5iQGxw5C"
|
||||
|
||||
|
||||
class ElevenLabsSynthesizer(BaseSynthesizer):
|
||||
def __init__(self, config: ElevenLabsSynthesizerConfig):
|
||||
super().__init__(config)
|
||||
self.api_key = config.api_key
|
||||
self.voice_id = config.voice_id or ADAM_VOICE_ID
|
||||
self.words_per_minute = 150
|
||||
|
||||
def create_speech(
|
||||
self,
|
||||
message: BaseMessage,
|
||||
chunk_size: int,
|
||||
bot_sentiment: Optional[BotSentiment] = None,
|
||||
) -> SynthesisResult:
|
||||
url = ELEVEN_LABS_BASE_URL + f"text-to-speech/{self.voice_id}/stream"
|
||||
headers = {"xi-api-key": self.api_key, "voice_id": self.voice_id}
|
||||
body = {
|
||||
"text": message.text,
|
||||
}
|
||||
response = requests.post(url, headers=headers, json=body)
|
||||
|
||||
def chunk_generator(response):
|
||||
for chunk in response.iter_content(chunk_size=chunk_size):
|
||||
yield SynthesisResult.ChunkResult(chunk, len(chunk) != chunk_size)
|
||||
|
||||
assert (
|
||||
not self.synthesizer_config.should_encode_as_wav
|
||||
), "ElevenLabs does not support WAV encoding"
|
||||
# return chunk_generator(response), lambda seconds: self.get_message_cutoff_from_voice_speed(message, seconds, self.words_per_minute)
|
||||
return SynthesisResult(chunk_generator(response), lambda seconds: message.text)
|
||||
BIN
vocode/streaming/synthesizer/filler_audio/typing-noise.wav
Normal file
BIN
vocode/streaming/synthesizer/filler_audio/typing-noise.wav
Normal file
Binary file not shown.
110
vocode/streaming/synthesizer/google_synthesizer.py
Normal file
110
vocode/streaming/synthesizer/google_synthesizer.py
Normal file
|
|
@ -0,0 +1,110 @@
|
|||
import io
|
||||
import wave
|
||||
from typing import Any, Optional
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from google.cloud import texttospeech_v1beta1 as tts
|
||||
|
||||
from vocode.streaming.agent.bot_sentiment_analyser import BotSentiment
|
||||
from vocode.streaming.models.message import BaseMessage
|
||||
from vocode.streaming.synthesizer.base_synthesizer import (
|
||||
BaseSynthesizer,
|
||||
SynthesisResult,
|
||||
encode_as_wav,
|
||||
)
|
||||
from vocode.streaming.models.synthesizer import GoogleSynthesizerConfig
|
||||
from vocode.streaming.models.audio_encoding import AudioEncoding
|
||||
from vocode.streaming.utils import convert_wav
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
||||
class GoogleSynthesizer(BaseSynthesizer):
|
||||
OFFSET_SECONDS = 0.5
|
||||
|
||||
def __init__(self, synthesizer_config: GoogleSynthesizerConfig):
|
||||
super().__init__(synthesizer_config)
|
||||
# Instantiates a client
|
||||
self.client = tts.TextToSpeechClient()
|
||||
|
||||
# Build the voice request, select the language code ("en-US") and the ssml
|
||||
# voice gender ("neutral")
|
||||
self.voice = tts.VoiceSelectionParams(
|
||||
language_code="en-US", name="en-US-Neural2-I"
|
||||
)
|
||||
|
||||
# Select the type of audio file you want returned
|
||||
self.audio_config = tts.AudioConfig(
|
||||
audio_encoding=tts.AudioEncoding.LINEAR16,
|
||||
sample_rate_hertz=24000,
|
||||
speaking_rate=1.2,
|
||||
pitch=0,
|
||||
effects_profile_id=["telephony-class-application"],
|
||||
)
|
||||
|
||||
def synthesize(self, message: str) -> tts.SynthesizeSpeechResponse:
|
||||
synthesis_input = tts.SynthesisInput(text=message)
|
||||
|
||||
# Perform the text-to-speech request on the text input with the selected
|
||||
# voice parameters and audio file type
|
||||
return self.client.synthesize_speech(
|
||||
request=tts.SynthesizeSpeechRequest(
|
||||
input=synthesis_input,
|
||||
voice=self.voice,
|
||||
audio_config=self.audio_config,
|
||||
enable_time_pointing=[
|
||||
tts.SynthesizeSpeechRequest.TimepointType.SSML_MARK
|
||||
],
|
||||
)
|
||||
)
|
||||
|
||||
def create_speech(
|
||||
self,
|
||||
message: BaseMessage,
|
||||
chunk_size: int,
|
||||
bot_sentiment: Optional[BotSentiment] = None,
|
||||
) -> SynthesisResult:
|
||||
response = self.synthesize(message.text)
|
||||
output_sample_rate = response.audio_config.sample_rate_hertz
|
||||
|
||||
real_offset = int(GoogleSynthesizer.OFFSET_SECONDS * output_sample_rate)
|
||||
|
||||
output_bytes_io = io.BytesIO()
|
||||
in_memory_wav = wave.open(output_bytes_io, "wb")
|
||||
in_memory_wav.setnchannels(1)
|
||||
in_memory_wav.setsampwidth(2)
|
||||
in_memory_wav.setframerate(output_sample_rate)
|
||||
in_memory_wav.writeframes(response.audio_content[real_offset:-real_offset])
|
||||
output_bytes_io.seek(0)
|
||||
|
||||
if self.synthesizer_config.audio_encoding == AudioEncoding.LINEAR16:
|
||||
output_bytes = convert_wav(
|
||||
output_bytes_io,
|
||||
output_sample_rate=self.synthesizer_config.sampling_rate,
|
||||
output_encoding=AudioEncoding.LINEAR16,
|
||||
)
|
||||
elif self.synthesizer_config.audio_encoding == AudioEncoding.MULAW:
|
||||
output_bytes = convert_wav(
|
||||
output_bytes_io,
|
||||
output_sample_rate=self.synthesizer_config.sampling_rate,
|
||||
output_encoding=AudioEncoding.MULAW,
|
||||
)
|
||||
|
||||
if self.synthesizer_config.should_encode_as_wav:
|
||||
output_bytes = encode_as_wav(output_bytes)
|
||||
|
||||
def chunk_generator(output_bytes):
|
||||
for i in range(0, len(output_bytes), chunk_size):
|
||||
if i + chunk_size > len(output_bytes):
|
||||
yield SynthesisResult.ChunkResult(output_bytes[i:], True)
|
||||
else:
|
||||
yield SynthesisResult.ChunkResult(
|
||||
output_bytes[i : i + chunk_size], False
|
||||
)
|
||||
|
||||
return SynthesisResult(
|
||||
chunk_generator(output_bytes),
|
||||
lambda seconds: self.get_message_cutoff_from_total_response_length(
|
||||
message, seconds, len(output_bytes)
|
||||
),
|
||||
)
|
||||
78
vocode/streaming/synthesizer/rime_synthesizer.py
Normal file
78
vocode/streaming/synthesizer/rime_synthesizer.py
Normal file
|
|
@ -0,0 +1,78 @@
|
|||
import audioop
|
||||
import base64
|
||||
from vocode.streaming.agent.bot_sentiment_analyser import BotSentiment
|
||||
from vocode.streaming.models.audio_encoding import AudioEncoding
|
||||
|
||||
from vocode.streaming.models.message import BaseMessage
|
||||
|
||||
from .base_synthesizer import BaseSynthesizer, SynthesisResult, encode_as_wav
|
||||
from typing import Any, Optional
|
||||
import os
|
||||
import io
|
||||
import wave
|
||||
from dotenv import load_dotenv
|
||||
import requests
|
||||
|
||||
from ..utils import convert_linear_audio, convert_wav
|
||||
from ..models.synthesizer import ElevenLabsSynthesizerConfig, RimeSynthesizerConfig
|
||||
|
||||
load_dotenv()
|
||||
|
||||
RIME_API_KEY = os.getenv("RIME_API_KEY")
|
||||
RIME_BASE_URL = os.getenv("RIME_BASE_URL")
|
||||
|
||||
|
||||
class RimeSynthesizer(BaseSynthesizer):
|
||||
def __init__(self, config: RimeSynthesizerConfig):
|
||||
super().__init__(config)
|
||||
self.speaker = config.speaker
|
||||
|
||||
def create_speech(
|
||||
self,
|
||||
message: BaseMessage,
|
||||
chunk_size: int,
|
||||
bot_sentiment: Optional[BotSentiment] = None,
|
||||
) -> SynthesisResult:
|
||||
url = RIME_BASE_URL
|
||||
headers = {"Authorization": f"Bearer {RIME_API_KEY}"}
|
||||
body = {"inputs": {"text": message.text, "speaker": self.speaker}}
|
||||
response = requests.post(url, headers=headers, json=body)
|
||||
|
||||
def chunk_generator(audio, chunk_transform=lambda x: x):
|
||||
for i in range(0, len(audio), chunk_size):
|
||||
chunk = audio[i : i + chunk_size]
|
||||
yield SynthesisResult.ChunkResult(
|
||||
chunk_transform(chunk), len(chunk) != chunk_size
|
||||
)
|
||||
|
||||
assert response.ok, response.text
|
||||
data = response.json().get("data")
|
||||
assert data
|
||||
|
||||
audio_file = io.BytesIO(base64.b64decode(data))
|
||||
|
||||
if self.synthesizer_config.audio_encoding == AudioEncoding.LINEAR16:
|
||||
output_bytes = convert_wav(
|
||||
audio_file,
|
||||
output_sample_rate=self.synthesizer_config.sampling_rate,
|
||||
output_encoding=AudioEncoding.LINEAR16,
|
||||
)
|
||||
elif self.synthesizer_config.audio_encoding == AudioEncoding.MULAW:
|
||||
output_bytes = convert_wav(
|
||||
audio_file,
|
||||
output_sample_rate=self.synthesizer_config.sampling_rate,
|
||||
output_encoding=AudioEncoding.MULAW,
|
||||
)
|
||||
|
||||
if self.synthesizer_config.should_encode_as_wav:
|
||||
output_generator = chunk_generator(
|
||||
output_bytes, chunk_transform=encode_as_wav
|
||||
)
|
||||
else:
|
||||
output_generator = chunk_generator(output_bytes)
|
||||
return SynthesisResult(
|
||||
output_generator,
|
||||
lambda seconds: self.get_message_cutoff_from_total_response_length(
|
||||
message, seconds, len(output_bytes)
|
||||
),
|
||||
)
|
||||
Loading…
Add table
Add a link
Reference in a new issue