remove pyq goodbye model and rime synthesizer and fix environment loading
This commit is contained in:
parent
a93bfc1ec9
commit
1dc7bc74c3
28 changed files with 143 additions and 285 deletions
|
|
@ -4,7 +4,7 @@ import re
|
|||
from typing import Any, Optional
|
||||
from xml.etree import ElementTree
|
||||
import azure.cognitiveservices.speech as speechsdk
|
||||
from dotenv import load_dotenv
|
||||
from vocode import getenv
|
||||
|
||||
from vocode.streaming.agent.bot_sentiment_analyser import BotSentiment
|
||||
from vocode.streaming.models.message import BaseMessage, SSMLMessage
|
||||
|
|
@ -20,7 +20,6 @@ from vocode.streaming.synthesizer.base_synthesizer import (
|
|||
from vocode.streaming.models.synthesizer import AzureSynthesizerConfig
|
||||
from vocode.streaming.models.audio_encoding import AudioEncoding
|
||||
|
||||
load_dotenv()
|
||||
|
||||
NAMESPACES = {
|
||||
"mstts": "https://www.w3.org/2001/mstts",
|
||||
|
|
@ -59,8 +58,8 @@ class AzureSynthesizer(BaseSynthesizer):
|
|||
self.synthesizer_config = synthesizer_config
|
||||
# Instantiates a client
|
||||
speech_config = speechsdk.SpeechConfig(
|
||||
subscription=os.environ.get("AZURE_SPEECH_KEY"),
|
||||
region=os.environ.get("AZURE_SPEECH_REGION"),
|
||||
subscription=getenv("AZURE_SPEECH_KEY"),
|
||||
region=getenv("AZURE_SPEECH_REGION"),
|
||||
)
|
||||
if self.synthesizer_config.audio_encoding == AudioEncoding.LINEAR16:
|
||||
if self.synthesizer_config.sampling_rate == 44100:
|
||||
|
|
|
|||
|
|
@ -1,7 +1,6 @@
|
|||
from typing import Any, Optional
|
||||
import os
|
||||
from dotenv import load_dotenv
|
||||
import requests
|
||||
from vocode import getenv
|
||||
|
||||
from vocode.streaming.synthesizer.base_synthesizer import (
|
||||
BaseSynthesizer,
|
||||
|
|
@ -11,9 +10,7 @@ from vocode.streaming.models.synthesizer import ElevenLabsSynthesizerConfig
|
|||
from vocode.streaming.agent.bot_sentiment_analyser import BotSentiment
|
||||
from vocode.streaming.models.message import BaseMessage
|
||||
|
||||
load_dotenv()
|
||||
|
||||
ELEVEN_LABS_API_KEY = os.environ.get("ELEVEN_LABS_API_KEY")
|
||||
ELEVEN_LABS_BASE_URL = "https://api.elevenlabs.io/v1/"
|
||||
ADAM_VOICE_ID = "pNInz6obpgDQGcFmaJgB"
|
||||
OBAMA_VOICE_ID = "vLITIS0SH2an5iQGxw5C"
|
||||
|
|
@ -22,7 +19,7 @@ OBAMA_VOICE_ID = "vLITIS0SH2an5iQGxw5C"
|
|||
class ElevenLabsSynthesizer(BaseSynthesizer):
|
||||
def __init__(self, config: ElevenLabsSynthesizerConfig):
|
||||
super().__init__(config)
|
||||
self.api_key = config.api_key
|
||||
self.api_key = getenv("ELEVEN_LABS_API_KEY")
|
||||
self.voice_id = config.voice_id or ADAM_VOICE_ID
|
||||
self.words_per_minute = 150
|
||||
|
||||
|
|
|
|||
|
|
@ -2,7 +2,6 @@ import io
|
|||
import wave
|
||||
from typing import Any, Optional
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from google.cloud import texttospeech_v1beta1 as tts
|
||||
|
||||
from vocode.streaming.agent.bot_sentiment_analyser import BotSentiment
|
||||
|
|
@ -16,8 +15,6 @@ from vocode.streaming.models.synthesizer import GoogleSynthesizerConfig
|
|||
from vocode.streaming.models.audio_encoding import AudioEncoding
|
||||
from vocode.streaming.utils import convert_wav
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
||||
class GoogleSynthesizer(BaseSynthesizer):
|
||||
OFFSET_SECONDS = 0.5
|
||||
|
|
|
|||
|
|
@ -1,78 +0,0 @@
|
|||
import audioop
|
||||
import base64
|
||||
from vocode.streaming.agent.bot_sentiment_analyser import BotSentiment
|
||||
from vocode.streaming.models.audio_encoding import AudioEncoding
|
||||
|
||||
from vocode.streaming.models.message import BaseMessage
|
||||
|
||||
from .base_synthesizer import BaseSynthesizer, SynthesisResult, encode_as_wav
|
||||
from typing import Any, Optional
|
||||
import os
|
||||
import io
|
||||
import wave
|
||||
from dotenv import load_dotenv
|
||||
import requests
|
||||
|
||||
from ..utils import convert_linear_audio, convert_wav
|
||||
from ..models.synthesizer import ElevenLabsSynthesizerConfig, RimeSynthesizerConfig
|
||||
|
||||
load_dotenv()
|
||||
|
||||
RIME_API_KEY = os.getenv("RIME_API_KEY")
|
||||
RIME_BASE_URL = os.getenv("RIME_BASE_URL")
|
||||
|
||||
|
||||
class RimeSynthesizer(BaseSynthesizer):
|
||||
def __init__(self, config: RimeSynthesizerConfig):
|
||||
super().__init__(config)
|
||||
self.speaker = config.speaker
|
||||
|
||||
def create_speech(
|
||||
self,
|
||||
message: BaseMessage,
|
||||
chunk_size: int,
|
||||
bot_sentiment: Optional[BotSentiment] = None,
|
||||
) -> SynthesisResult:
|
||||
url = RIME_BASE_URL
|
||||
headers = {"Authorization": f"Bearer {RIME_API_KEY}"}
|
||||
body = {"inputs": {"text": message.text, "speaker": self.speaker}}
|
||||
response = requests.post(url, headers=headers, json=body)
|
||||
|
||||
def chunk_generator(audio, chunk_transform=lambda x: x):
|
||||
for i in range(0, len(audio), chunk_size):
|
||||
chunk = audio[i : i + chunk_size]
|
||||
yield SynthesisResult.ChunkResult(
|
||||
chunk_transform(chunk), len(chunk) != chunk_size
|
||||
)
|
||||
|
||||
assert response.ok, response.text
|
||||
data = response.json().get("data")
|
||||
assert data
|
||||
|
||||
audio_file = io.BytesIO(base64.b64decode(data))
|
||||
|
||||
if self.synthesizer_config.audio_encoding == AudioEncoding.LINEAR16:
|
||||
output_bytes = convert_wav(
|
||||
audio_file,
|
||||
output_sample_rate=self.synthesizer_config.sampling_rate,
|
||||
output_encoding=AudioEncoding.LINEAR16,
|
||||
)
|
||||
elif self.synthesizer_config.audio_encoding == AudioEncoding.MULAW:
|
||||
output_bytes = convert_wav(
|
||||
audio_file,
|
||||
output_sample_rate=self.synthesizer_config.sampling_rate,
|
||||
output_encoding=AudioEncoding.MULAW,
|
||||
)
|
||||
|
||||
if self.synthesizer_config.should_encode_as_wav:
|
||||
output_generator = chunk_generator(
|
||||
output_bytes, chunk_transform=encode_as_wav
|
||||
)
|
||||
else:
|
||||
output_generator = chunk_generator(output_bytes)
|
||||
return SynthesisResult(
|
||||
output_generator,
|
||||
lambda seconds: self.get_message_cutoff_from_total_response_length(
|
||||
message, seconds, len(output_bytes)
|
||||
),
|
||||
)
|
||||
Loading…
Add table
Add a link
Reference in a new issue