Added support for Google Journey TTS voices

This commit is contained in:
Mark Backman
2025-01-06 14:54:34 -05:00
parent a1377b7f1a
commit 3e1ec4a8ee
3 changed files with 17 additions and 7 deletions

View File

@@ -7,10 +7,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
## [Unreleased] ## [Unreleased]
- Added new foundational examples for `LiveKitTransportLayer` - ### Added
- `29-livekit-audio-chat.py` - Supports both Audio to Audio and Text to Audio - Added support for Google TTS Journey voices in `GoogleTTSService`.
chat pipelines.
- Added `29-livekit-audio-chat.py`, as a new foundational examples for
`LiveKitTransportLayer`.
## [0.0.52] - 2024-12-24 ## [0.0.52] - 2024-12-24

View File

@@ -22,6 +22,7 @@ from pipecat.processors.aggregators.openai_llm_context import OpenAILLMContext
from pipecat.services.deepgram import DeepgramSTTService from pipecat.services.deepgram import DeepgramSTTService
from pipecat.services.google import GoogleTTSService from pipecat.services.google import GoogleTTSService
from pipecat.services.openai import OpenAILLMService from pipecat.services.openai import OpenAILLMService
from pipecat.transcriptions.language import Language
from pipecat.transports.services.daily import DailyParams, DailyTransport from pipecat.transports.services.daily import DailyParams, DailyTransport
load_dotenv(override=True) load_dotenv(override=True)
@@ -50,8 +51,8 @@ async def main():
stt = DeepgramSTTService(api_key=os.getenv("DEEPGRAM_API_KEY")) stt = DeepgramSTTService(api_key=os.getenv("DEEPGRAM_API_KEY"))
tts = GoogleTTSService( tts = GoogleTTSService(
voice_id="en-US-Neural2-J", voice_id="en-US-Journey-F",
params=GoogleTTSService.InputParams(language="en-US", rate="1.05"), params=GoogleTTSService.InputParams(language=Language.EN_US),
) )
llm = OpenAILLMService(api_key=os.getenv("OPENAI_API_KEY"), model="gpt-4o") llm = OpenAILLMService(api_key=os.getenv("OPENAI_API_KEY"), model="gpt-4o")

View File

@@ -865,8 +865,15 @@ class GoogleTTSService(TTSService):
try: try:
await self.start_ttfb_metrics() await self.start_ttfb_metrics()
ssml = self._construct_ssml(text) is_journey_voice = "journey" in self._voice_id.lower()
synthesis_input = texttospeech_v1.SynthesisInput(ssml=ssml)
# Create synthesis input based on voice_id
if is_journey_voice:
synthesis_input = texttospeech_v1.SynthesisInput(text=text)
else:
ssml = self._construct_ssml(text)
synthesis_input = texttospeech_v1.SynthesisInput(ssml=ssml)
voice = texttospeech_v1.VoiceSelectionParams( voice = texttospeech_v1.VoiceSelectionParams(
language_code=self._settings["language"], name=self._voice_id language_code=self._settings["language"], name=self._voice_id
) )