Gradium integration.
This commit is contained in:
@@ -63,6 +63,7 @@ fireworks = []
|
|||||||
fish = [ "ormsgpack~=1.7.0", "pipecat-ai[websockets-base]" ]
|
fish = [ "ormsgpack~=1.7.0", "pipecat-ai[websockets-base]" ]
|
||||||
gladia = [ "pipecat-ai[websockets-base]" ]
|
gladia = [ "pipecat-ai[websockets-base]" ]
|
||||||
google = [ "google-cloud-speech>=2.33.0,<3", "google-cloud-texttospeech>=2.31.0,<3", "google-genai>=1.41.0,<2", "pipecat-ai[websockets-base]" ]
|
google = [ "google-cloud-speech>=2.33.0,<3", "google-cloud-texttospeech>=2.31.0,<3", "google-genai>=1.41.0,<2", "pipecat-ai[websockets-base]" ]
|
||||||
|
gradium = [ "pipecat-ai[websockets-base]" ]
|
||||||
grok = []
|
grok = []
|
||||||
groq = [ "groq~=0.23.0" ]
|
groq = [ "groq~=0.23.0" ]
|
||||||
gstreamer = [ "pygobject~=3.50.0" ]
|
gstreamer = [ "pygobject~=3.50.0" ]
|
||||||
|
|||||||
5
src/pipecat/services/gradium/__init__.py
Normal file
5
src/pipecat/services/gradium/__init__.py
Normal file
@@ -0,0 +1,5 @@
|
|||||||
|
#
|
||||||
|
# Copyright (c) 2024–2025, Daily
|
||||||
|
#
|
||||||
|
# SPDX-License-Identifier: BSD 2-Clause License
|
||||||
|
#
|
||||||
256
src/pipecat/services/gradium/stt.py
Normal file
256
src/pipecat/services/gradium/stt.py
Normal file
@@ -0,0 +1,256 @@
|
|||||||
|
#
|
||||||
|
# Copyright (c) 2024–2025, Daily
|
||||||
|
#
|
||||||
|
# SPDX-License-Identifier: BSD 2-Clause License
|
||||||
|
#
|
||||||
|
|
||||||
|
"""Gradium's speech-to-text service implementation.
|
||||||
|
|
||||||
|
This module provides integration with Gradium's real-time speech-to-text
|
||||||
|
WebSocket API for streaming audio transcription.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import base64
|
||||||
|
import json
|
||||||
|
from typing import AsyncGenerator
|
||||||
|
|
||||||
|
from loguru import logger
|
||||||
|
|
||||||
|
from pipecat.frames.frames import (
|
||||||
|
CancelFrame,
|
||||||
|
EndFrame,
|
||||||
|
Frame,
|
||||||
|
StartFrame,
|
||||||
|
TranscriptionFrame,
|
||||||
|
UserStartedSpeakingFrame,
|
||||||
|
UserStoppedSpeakingFrame,
|
||||||
|
)
|
||||||
|
from pipecat.processors.frame_processor import FrameDirection
|
||||||
|
from pipecat.services.stt_service import WebsocketSTTService
|
||||||
|
from pipecat.transcriptions.language import Language
|
||||||
|
from pipecat.utils.time import time_now_iso8601
|
||||||
|
from pipecat.utils.tracing.service_decorators import traced_stt
|
||||||
|
|
||||||
|
try:
|
||||||
|
from websockets.asyncio.client import connect as websocket_connect
|
||||||
|
from websockets.protocol import State
|
||||||
|
except ModuleNotFoundError as e:
|
||||||
|
logger.error(f"Exception: {e}")
|
||||||
|
logger.error('In order to use Gradium, you need to `pip install "pipecat-ai[gradium]"`.')
|
||||||
|
raise Exception(f"Missing module: {e}")
|
||||||
|
|
||||||
|
SAMPLE_RATE = 24000
|
||||||
|
|
||||||
|
|
||||||
|
class GradiumSTTService(WebsocketSTTService):
|
||||||
|
"""Gradium real-time speech-to-text service.
|
||||||
|
|
||||||
|
Provides real-time speech transcription using Gradium's WebSocket API.
|
||||||
|
Supports both interim and final transcriptions with configurable parameters
|
||||||
|
for audio processing and connection management.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
api_key: str,
|
||||||
|
api_endpoint_base_url: str = "wss://eu.api.gradium.ai/api/speech/asr",
|
||||||
|
json_config: str | None = None,
|
||||||
|
**kwargs,
|
||||||
|
):
|
||||||
|
"""Initialize the Gradium STT service.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
api_key: Gradium API key for authentication.
|
||||||
|
language: Language code for transcription. Defaults to English (Language.EN).
|
||||||
|
api_endpoint_base_url: WebSocket endpoint URL. Defaults to Gradium's streaming endpoint.
|
||||||
|
json_config: Optional JSON configuration string for additional model settings.
|
||||||
|
**kwargs: Additional arguments passed to parent STTService class.
|
||||||
|
"""
|
||||||
|
super().__init__(sample_rate=SAMPLE_RATE, **kwargs)
|
||||||
|
|
||||||
|
self._api_key = api_key
|
||||||
|
self._api_endpoint_base_url = api_endpoint_base_url
|
||||||
|
self._websocket = None
|
||||||
|
self._json_config = json_config
|
||||||
|
|
||||||
|
self._receive_task = None
|
||||||
|
|
||||||
|
self._audio_buffer = bytearray()
|
||||||
|
self._chunk_size_ms = 80
|
||||||
|
self._chunk_size_bytes = 0
|
||||||
|
|
||||||
|
def can_generate_metrics(self) -> bool:
|
||||||
|
"""Check if the service can generate metrics.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if metrics generation is supported.
|
||||||
|
"""
|
||||||
|
return True
|
||||||
|
|
||||||
|
async def start(self, frame: StartFrame):
|
||||||
|
"""Start the speech-to-text service.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
frame: Start frame to begin processing.
|
||||||
|
"""
|
||||||
|
await super().start(frame)
|
||||||
|
self._chunk_size_bytes = int(self._chunk_size_ms * self._sample_rate * 2 / 1000)
|
||||||
|
await self._connect()
|
||||||
|
|
||||||
|
async def stop(self, frame: EndFrame):
|
||||||
|
"""Stop the speech-to-text service.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
frame: End frame to stop processing.
|
||||||
|
"""
|
||||||
|
await super().stop(frame)
|
||||||
|
await self._disconnect()
|
||||||
|
|
||||||
|
async def cancel(self, frame: CancelFrame):
|
||||||
|
"""Cancel the speech-to-text service.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
frame: Cancel frame to abort processing.
|
||||||
|
"""
|
||||||
|
await super().cancel(frame)
|
||||||
|
await self._disconnect()
|
||||||
|
|
||||||
|
async def run_stt(self, audio: bytes) -> AsyncGenerator[Frame, None]:
|
||||||
|
"""Process audio data for speech-to-text conversion.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
audio: Raw audio bytes to process.
|
||||||
|
|
||||||
|
Yields:
|
||||||
|
None (processing handled via WebSocket messages).
|
||||||
|
"""
|
||||||
|
self._audio_buffer.extend(audio)
|
||||||
|
|
||||||
|
while len(self._audio_buffer) >= self._chunk_size_bytes:
|
||||||
|
chunk = bytes(self._audio_buffer[: self._chunk_size_bytes])
|
||||||
|
self._audio_buffer = self._audio_buffer[self._chunk_size_bytes :]
|
||||||
|
chunk = base64.b64encode(chunk).decode("utf-8")
|
||||||
|
msg = {"type": "audio", "audio": chunk}
|
||||||
|
if self._websocket and self._websocket.state is State.OPEN:
|
||||||
|
await self._websocket.send(json.dumps(msg))
|
||||||
|
|
||||||
|
yield None
|
||||||
|
|
||||||
|
async def process_frame(self, frame: Frame, direction: FrameDirection):
|
||||||
|
"""Process frames for VAD and metrics handling.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
frame: Frame to process.
|
||||||
|
direction: Direction of frame processing.
|
||||||
|
"""
|
||||||
|
await super().process_frame(frame, direction)
|
||||||
|
if isinstance(frame, UserStartedSpeakingFrame):
|
||||||
|
await self.start_ttfb_metrics()
|
||||||
|
elif isinstance(frame, UserStoppedSpeakingFrame):
|
||||||
|
await self.start_processing_metrics()
|
||||||
|
|
||||||
|
@traced_stt
|
||||||
|
async def _trace_transcription(self, transcript: str, is_final: bool, language: Language):
|
||||||
|
"""Record transcription event for tracing."""
|
||||||
|
pass
|
||||||
|
|
||||||
|
async def _connect(self):
|
||||||
|
await self._connect_websocket()
|
||||||
|
|
||||||
|
if self._websocket and not self._receive_task:
|
||||||
|
self._receive_task = self.create_task(self._receive_task_handler(self._report_error))
|
||||||
|
|
||||||
|
async def _connect_websocket(self):
|
||||||
|
try:
|
||||||
|
if self._websocket and self._websocket.state is State.OPEN:
|
||||||
|
return
|
||||||
|
ws_url = self._api_endpoint_base_url
|
||||||
|
headers = {
|
||||||
|
"x-api-key": self._api_key,
|
||||||
|
"x-api-source": "pipecat",
|
||||||
|
}
|
||||||
|
self._websocket = await websocket_connect(
|
||||||
|
ws_url,
|
||||||
|
additional_headers=headers,
|
||||||
|
)
|
||||||
|
await self._call_event_handler("on_connected")
|
||||||
|
setup_msg = {
|
||||||
|
"type": "setup",
|
||||||
|
"input_format": "pcm",
|
||||||
|
}
|
||||||
|
if self._json_config is not None:
|
||||||
|
setup_msg["json_config"] = self._json_config
|
||||||
|
await self._websocket.send(json.dumps(setup_msg))
|
||||||
|
ready_msg = await self._websocket.recv()
|
||||||
|
ready_msg = json.loads(ready_msg)
|
||||||
|
if ready_msg["type"] == "error":
|
||||||
|
raise Exception(f"received error {ready_msg['message']}")
|
||||||
|
if ready_msg["type"] != "ready":
|
||||||
|
raise Exception(f"unexpected first message type {ready_msg['type']}")
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"{self}: unable to connect to Gradium: {e}")
|
||||||
|
raise
|
||||||
|
|
||||||
|
async def _disconnect(self):
|
||||||
|
if self._receive_task:
|
||||||
|
await self.cancel_task(self._receive_task)
|
||||||
|
self._receive_task = None
|
||||||
|
|
||||||
|
await self._disconnect_websocket()
|
||||||
|
|
||||||
|
async def _disconnect_websocket(self):
|
||||||
|
try:
|
||||||
|
if self._websocket and self._websocket.state is State.OPEN:
|
||||||
|
logger.debug("Disconnecting from Gradium STT")
|
||||||
|
await self._websocket.close()
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"{self} error closing websocket: {e}")
|
||||||
|
finally:
|
||||||
|
self._websocket = None
|
||||||
|
await self._call_event_handler("on_disconnected")
|
||||||
|
|
||||||
|
def _get_websocket(self):
|
||||||
|
if self._websocket:
|
||||||
|
return self._websocket
|
||||||
|
raise Exception("Websocket not connected")
|
||||||
|
|
||||||
|
async def _process_messages(self):
|
||||||
|
async for message in self._get_websocket():
|
||||||
|
try:
|
||||||
|
data = json.loads(message)
|
||||||
|
await self._process_response(data)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
logger.warning(f"Received non-JSON message: {message}")
|
||||||
|
|
||||||
|
async def _receive_messages(self):
|
||||||
|
while True:
|
||||||
|
await self._process_messages()
|
||||||
|
logger.debug(f"{self} Gradium connection was disconnected (timeout?), reconnecting")
|
||||||
|
await self._connect_websocket()
|
||||||
|
|
||||||
|
async def _process_response(self, msg):
|
||||||
|
type_ = msg.get("type", "")
|
||||||
|
if type_ == "text":
|
||||||
|
await self._handle_text(msg["text"])
|
||||||
|
elif type_ == "end_of_stream":
|
||||||
|
await self._handle_end_of_stream()
|
||||||
|
elif type_ == "error":
|
||||||
|
logger.error(f"Gradium error: {msg.get('message', 'Unknown error')}")
|
||||||
|
|
||||||
|
async def _handle_end_of_stream(self):
|
||||||
|
"""Handle termination message."""
|
||||||
|
logger.info("Received end_of_stream message from server")
|
||||||
|
await self.push_frame(EndFrame())
|
||||||
|
|
||||||
|
async def _handle_text(self, text: str):
|
||||||
|
"""Handle transcription results."""
|
||||||
|
await self.push_frame(
|
||||||
|
TranscriptionFrame(
|
||||||
|
text,
|
||||||
|
self._user_id,
|
||||||
|
time_now_iso8601(),
|
||||||
|
)
|
||||||
|
)
|
||||||
338
src/pipecat/services/gradium/tts.py
Normal file
338
src/pipecat/services/gradium/tts.py
Normal file
@@ -0,0 +1,338 @@
|
|||||||
|
# Copyright (c) 2024–2025, Daily
|
||||||
|
#
|
||||||
|
# SPDX-License-Identifier: BSD 2-Clause License
|
||||||
|
|
||||||
|
"""Gradium Text-to-Speech service implementation."""
|
||||||
|
|
||||||
|
import base64
|
||||||
|
import json
|
||||||
|
import uuid
|
||||||
|
from typing import Any, AsyncGenerator, Mapping, Optional
|
||||||
|
|
||||||
|
from loguru import logger
|
||||||
|
from pydantic import BaseModel
|
||||||
|
|
||||||
|
from pipecat.frames.frames import (
|
||||||
|
CancelFrame,
|
||||||
|
EndFrame,
|
||||||
|
ErrorFrame,
|
||||||
|
Frame,
|
||||||
|
InterruptionFrame,
|
||||||
|
StartFrame,
|
||||||
|
TTSAudioRawFrame,
|
||||||
|
TTSStartedFrame,
|
||||||
|
TTSStoppedFrame,
|
||||||
|
)
|
||||||
|
from pipecat.processors.frame_processor import FrameDirection
|
||||||
|
from pipecat.services.tts_service import InterruptibleWordTTSService
|
||||||
|
from pipecat.utils.tracing.service_decorators import traced_tts
|
||||||
|
|
||||||
|
try:
|
||||||
|
from websockets import ConnectionClosedOK
|
||||||
|
from websockets.asyncio.client import connect as websocket_connect
|
||||||
|
from websockets.protocol import State
|
||||||
|
except ModuleNotFoundError as e:
|
||||||
|
logger.error(f"Exception: {e}")
|
||||||
|
logger.error("In order to use Gradium, you need to `pip install pipecat-ai[gradium]`.")
|
||||||
|
raise Exception(f"Missing module: {e}")
|
||||||
|
|
||||||
|
SAMPLE_RATE = 48000
|
||||||
|
|
||||||
|
|
||||||
|
class GradiumTTSService(InterruptibleWordTTSService):
|
||||||
|
"""Text-to-Speech service using Gradium's websocket API."""
|
||||||
|
|
||||||
|
class InputParams(BaseModel):
|
||||||
|
"""Configuration parameters for Gradium TTS service.
|
||||||
|
|
||||||
|
Parameters:
|
||||||
|
temp: Temperature to be used for generation, defaults to 0.6.
|
||||||
|
"""
|
||||||
|
|
||||||
|
temp: Optional[float] = 0.6
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
api_key: str,
|
||||||
|
voice: Optional[str] = None,
|
||||||
|
voice_id: Optional[str] = None,
|
||||||
|
url: str = "wss://eu.api.gradium.ai/api/speech/tts",
|
||||||
|
model: str = "default",
|
||||||
|
json_config: Optional[str] = None,
|
||||||
|
params: Optional[InputParams] = None,
|
||||||
|
**kwargs,
|
||||||
|
):
|
||||||
|
"""Initialize the Gradium TTS service.
|
||||||
|
|
||||||
|
The voice used in the generation is specified by setting either the `voice` argument
|
||||||
|
for catalog voices or the `voice_id` argument for custom voices.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
api_key: Gradium API key for authentication.
|
||||||
|
voice: name for a catalog voice.
|
||||||
|
voice_id: identifier for a custom voice.
|
||||||
|
url: Gradium websocket API endpoint.
|
||||||
|
model: Model ID to use for synthesis.
|
||||||
|
json_config: Optional JSON configuration string for additional model settings.
|
||||||
|
params: Additional configuration parameters.
|
||||||
|
**kwargs: Additional arguments passed to parent class.
|
||||||
|
"""
|
||||||
|
# Initialize with parent class settings for proper frame handling
|
||||||
|
super().__init__(
|
||||||
|
push_stop_frames=True,
|
||||||
|
pause_frame_processing=True,
|
||||||
|
sample_rate=SAMPLE_RATE,
|
||||||
|
**kwargs,
|
||||||
|
)
|
||||||
|
|
||||||
|
if voice is None and voice_id is None:
|
||||||
|
raise ValueError("one of voice and voice_id has to be set")
|
||||||
|
if voice is not None and voice_id is not None:
|
||||||
|
raise ValueError("voice and voice_id cannot be set at the same time")
|
||||||
|
|
||||||
|
params = params or GradiumTTSService.InputParams()
|
||||||
|
|
||||||
|
# Store service configuration
|
||||||
|
self._api_key = api_key
|
||||||
|
self._url = url
|
||||||
|
self._voice = voice
|
||||||
|
self._voice_id = voice_id
|
||||||
|
self._json_config = json_config
|
||||||
|
self._model = model
|
||||||
|
self._settings = {
|
||||||
|
"voice": voice,
|
||||||
|
"voice_id": voice_id,
|
||||||
|
"model_name": model,
|
||||||
|
"output_format": "pcm",
|
||||||
|
}
|
||||||
|
|
||||||
|
# State tracking
|
||||||
|
self._receive_task = None
|
||||||
|
self._request_id = None
|
||||||
|
|
||||||
|
def can_generate_metrics(self) -> bool:
|
||||||
|
"""Check if this service can generate processing metrics.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True, as Gradium service supports metrics generation.
|
||||||
|
"""
|
||||||
|
return True
|
||||||
|
|
||||||
|
async def set_model(self, model: str):
|
||||||
|
"""Update the TTS model.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
model: The model name to use for synthesis.
|
||||||
|
"""
|
||||||
|
self._model = model
|
||||||
|
await super().set_model(model)
|
||||||
|
|
||||||
|
async def _update_settings(self, settings: Mapping[str, Any]):
|
||||||
|
"""Update service settings and reconnect if voice changed."""
|
||||||
|
prev_voice = self._voice_id
|
||||||
|
await super()._update_settings(settings)
|
||||||
|
if not prev_voice == self._voice_id:
|
||||||
|
self._settings["voice_id"] = self._voice_id
|
||||||
|
logger.info(f"Switching TTS voice to: [{self._voice_id}]")
|
||||||
|
await self._disconnect()
|
||||||
|
await self._connect()
|
||||||
|
|
||||||
|
def _build_msg(self, text: str = "") -> dict:
|
||||||
|
"""Build JSON message for Gradium API."""
|
||||||
|
return {"text": text, "type": "text"}
|
||||||
|
|
||||||
|
async def start(self, frame: StartFrame):
|
||||||
|
"""Start the service and establish websocket connection.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
frame: The start frame containing initialization parameters.
|
||||||
|
"""
|
||||||
|
await super().start(frame)
|
||||||
|
await self._connect()
|
||||||
|
|
||||||
|
async def stop(self, frame: EndFrame):
|
||||||
|
"""Stop the service and close connection.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
frame: The end frame.
|
||||||
|
"""
|
||||||
|
await super().stop(frame)
|
||||||
|
await self._disconnect()
|
||||||
|
|
||||||
|
async def cancel(self, frame: CancelFrame):
|
||||||
|
"""Cancel current operation and clean up.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
frame: The cancel frame.
|
||||||
|
"""
|
||||||
|
await super().cancel(frame)
|
||||||
|
await self._disconnect()
|
||||||
|
|
||||||
|
async def _connect(self):
|
||||||
|
"""Establish websocket connection and start receive task."""
|
||||||
|
logger.debug(f"{self}: connecting")
|
||||||
|
|
||||||
|
# If the server disconnected, cancel the receive-task so that it can be reset below.
|
||||||
|
if self._websocket is None or self._websocket.state is not State.OPEN:
|
||||||
|
if self._receive_task:
|
||||||
|
await self.cancel_task(self._receive_task)
|
||||||
|
self._receive_task = None
|
||||||
|
|
||||||
|
await self._connect_websocket()
|
||||||
|
|
||||||
|
if self._websocket and not self._receive_task:
|
||||||
|
logger.debug(f"{self}: setting receive task")
|
||||||
|
self._receive_task = self.create_task(self._receive_task_handler(self._report_error))
|
||||||
|
|
||||||
|
async def _disconnect(self):
|
||||||
|
"""Close websocket connection and clean up tasks."""
|
||||||
|
logger.debug(f"{self}: disconnecting")
|
||||||
|
if self._receive_task:
|
||||||
|
await self.cancel_task(self._receive_task)
|
||||||
|
self._receive_task = None
|
||||||
|
|
||||||
|
await self._disconnect_websocket()
|
||||||
|
|
||||||
|
async def _connect_websocket(self):
|
||||||
|
"""Connect to Gradium websocket API with configured settings."""
|
||||||
|
try:
|
||||||
|
if self._websocket and self._websocket.state is State.OPEN:
|
||||||
|
return
|
||||||
|
|
||||||
|
headers = {"x-api-key": self._api_key, "x-api-source": "pipecat"}
|
||||||
|
self._websocket = await websocket_connect(self._url, additional_headers=headers)
|
||||||
|
|
||||||
|
setup_msg = {
|
||||||
|
"type": "setup",
|
||||||
|
"output_format": "pcm",
|
||||||
|
}
|
||||||
|
if self._json_config is not None:
|
||||||
|
setup_msg["json_config"] = self._json_config
|
||||||
|
if self._voice is not None:
|
||||||
|
setup_msg["voice"] = self._voice
|
||||||
|
if self._voice_id is not None:
|
||||||
|
setup_msg["voice_id"] = self._voice_id
|
||||||
|
await self._websocket.send(json.dumps(setup_msg))
|
||||||
|
ready_msg = await self._websocket.recv()
|
||||||
|
ready_msg = json.loads(ready_msg)
|
||||||
|
if ready_msg["type"] == "error":
|
||||||
|
raise Exception(f"received error {ready_msg['message']}")
|
||||||
|
if ready_msg["type"] != "ready":
|
||||||
|
raise Exception(f"unexpected first message type {ready_msg['type']}")
|
||||||
|
|
||||||
|
await self._call_event_handler("on_connected")
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"{self} initialization error: {e}")
|
||||||
|
self._websocket = None
|
||||||
|
await self._call_event_handler("on_connection_error", f"{e}")
|
||||||
|
|
||||||
|
async def _disconnect_websocket(self):
|
||||||
|
"""Close websocket connection and reset state."""
|
||||||
|
try:
|
||||||
|
await self.stop_all_metrics()
|
||||||
|
if self._websocket:
|
||||||
|
await self._websocket.close()
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"{self} error closing websocket: {e}")
|
||||||
|
finally:
|
||||||
|
self._websocket = None
|
||||||
|
self._request_id = None
|
||||||
|
await self._call_event_handler("on_disconnected")
|
||||||
|
|
||||||
|
def _get_websocket(self):
|
||||||
|
"""Get active websocket connection or raise exception."""
|
||||||
|
if self._websocket:
|
||||||
|
return self._websocket
|
||||||
|
raise Exception("Websocket not connected")
|
||||||
|
|
||||||
|
async def flush_audio(self):
|
||||||
|
"""Flush any pending audio synthesis."""
|
||||||
|
if not self._websocket:
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
msg = {"type": "end_of_stream"}
|
||||||
|
await self._websocket.send(json.dumps(msg))
|
||||||
|
logger.trace(f"{self}: flushing audio")
|
||||||
|
except ConnectionClosedOK:
|
||||||
|
logger.debug(f"{self}: connection closed normally during flush")
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"{self} exception: {e}")
|
||||||
|
|
||||||
|
async def _receive_messages(self):
|
||||||
|
"""Process incoming websocket messages."""
|
||||||
|
# TODO(laurent): This should not be necessary as it should happen when
|
||||||
|
# receiving the messages but this does not seem to always be the case
|
||||||
|
# and that may lead to a busy polling loop.
|
||||||
|
if self._websocket and self._websocket.state is State.CLOSED:
|
||||||
|
raise ConnectionClosedOK(None, None)
|
||||||
|
async for message in self._get_websocket():
|
||||||
|
msg = json.loads(message)
|
||||||
|
|
||||||
|
if msg["type"] == "audio":
|
||||||
|
# Process audio chunk
|
||||||
|
await self.stop_ttfb_metrics()
|
||||||
|
self.start_word_timestamps()
|
||||||
|
frame = TTSAudioRawFrame(
|
||||||
|
audio=base64.b64decode(msg["audio"]),
|
||||||
|
sample_rate=self.sample_rate,
|
||||||
|
num_channels=1,
|
||||||
|
)
|
||||||
|
await self.push_frame(frame)
|
||||||
|
|
||||||
|
elif msg["type"] == "text":
|
||||||
|
await self.add_word_timestamps([(msg["text"], msg["start_s"])])
|
||||||
|
elif msg["type"] == "end_of_stream":
|
||||||
|
await self.push_frame(TTSStoppedFrame())
|
||||||
|
await self.stop_all_metrics()
|
||||||
|
|
||||||
|
elif msg["type"] == "error":
|
||||||
|
logger.error(f"{self} error: {msg}")
|
||||||
|
await self.push_frame(TTSStoppedFrame())
|
||||||
|
await self.stop_all_metrics()
|
||||||
|
await self.push_error(ErrorFrame(f"{self} error: {msg['message']}"))
|
||||||
|
|
||||||
|
async def push_frame(self, frame: Frame, direction: FrameDirection = FrameDirection.DOWNSTREAM):
|
||||||
|
"""Push frame and handle end-of-turn conditions.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
frame: The frame to push.
|
||||||
|
direction: The direction to push the frame.
|
||||||
|
"""
|
||||||
|
await super().push_frame(frame, direction)
|
||||||
|
|
||||||
|
@traced_tts
|
||||||
|
async def run_tts(self, text: str) -> AsyncGenerator[Frame, None]:
|
||||||
|
"""Generate speech from text using Gradium's streaming API.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
text: The text to convert to speech.
|
||||||
|
|
||||||
|
Yields:
|
||||||
|
Frame: Audio frames containing the synthesized speech.
|
||||||
|
"""
|
||||||
|
_state = self._websocket.state if self._websocket is not None else None
|
||||||
|
logger.debug(f"{self}: Generating TTS [{text}] {_state}")
|
||||||
|
try:
|
||||||
|
if not self._websocket or self._websocket.state is State.CLOSED:
|
||||||
|
self._websocket = None
|
||||||
|
await self._connect()
|
||||||
|
|
||||||
|
try:
|
||||||
|
if not self._request_id:
|
||||||
|
await self.start_ttfb_metrics()
|
||||||
|
yield TTSStartedFrame()
|
||||||
|
self._request_id = str(uuid.uuid4())
|
||||||
|
|
||||||
|
msg = self._build_msg(text=text)
|
||||||
|
await self._get_websocket().send(json.dumps(msg))
|
||||||
|
await self.start_tts_usage_metrics(text)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"{self} error sending message: {e}")
|
||||||
|
yield TTSStoppedFrame()
|
||||||
|
await self._disconnect()
|
||||||
|
await self._connect()
|
||||||
|
return
|
||||||
|
yield None
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"{self} exception: {e}")
|
||||||
Reference in New Issue
Block a user