Merge pull request #1593 from WebinarGeek/wg/gladia-translations

Push gladia translations as a TranscriptionFrame
This commit is contained in:
Mark Backman
2025-04-25 08:35:36 -04:00
committed by GitHub
5 changed files with 139 additions and 2 deletions

View File

@@ -9,6 +9,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added ### Added
- Added `TranslationFrame`, a new frame type that contains a translated
transcription.
- Added `TransportParams.audio_in_passthrough`. If set (the default), incoming - Added `TransportParams.audio_in_passthrough`. If set (the default), incoming
audio will be pushed downstream. audio will be pushed downstream.
@@ -17,6 +20,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Changed ### Changed
- Updated `GladiaSTTService` to output a `TranslationFrame` when specifying a
`translation` and `translation_config`.
- STT services now passthrough audio frames by default. This allows you to add - STT services now passthrough audio frames by default. This allows you to add
audio recording without worrying about what's wrong in your pipeline when it audio recording without worrying about what's wrong in your pipeline when it
doesn't work the first time. doesn't work the first time.
@@ -49,6 +55,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- Added 04 foundational examples for client/server transports. Also, renamed - Added 04 foundational examples for client/server transports. Also, renamed
`29-livekit-audio-chat.py` to `04b-transports-livekit.py`. `29-livekit-audio-chat.py` to `04b-transports-livekit.py`.
- Added foundational example `13c-gladia-translation.py` showing how to use
`TranscriptionFrame` and `TranslationFrame`.
## [0.0.65] - 2025-04-23 "Sant Jordi's release" 🌹📕 ## [0.0.65] - 2025-04-23 "Sant Jordi's release" 🌹📕
https://en.wikipedia.org/wiki/Saint_George%27s_Day_in_Catalonia https://en.wikipedia.org/wiki/Saint_George%27s_Day_in_Catalonia

View File

@@ -130,6 +130,12 @@ pip install "pipecat-ai[option,...]"
### Running tests ### Running tests
Install the test dependencies:
```shell
pip install -r test-requirements.txt
```
From the root directory, run: From the root directory, run:
```shell ```shell

View File

@@ -0,0 +1,90 @@
#
# Copyright (c) 20242025, Daily
#
# SPDX-License-Identifier: BSD 2-Clause License
#
import os
from dotenv import load_dotenv
from loguru import logger
from pipecat.frames.frames import Frame, TranscriptionFrame, TranslationFrame
from pipecat.pipeline.pipeline import Pipeline
from pipecat.pipeline.runner import PipelineRunner
from pipecat.pipeline.task import PipelineTask
from pipecat.processors.frame_processor import FrameDirection, FrameProcessor
from pipecat.services.gladia.config import (
GladiaInputParams,
LanguageConfig,
RealtimeProcessingConfig,
TranslationConfig,
)
from pipecat.services.gladia.stt import GladiaSTTService
from pipecat.transcriptions.language import Language
from pipecat.transports.base_transport import TransportParams
from pipecat.transports.network.small_webrtc import SmallWebRTCTransport
from pipecat.transports.network.webrtc_connection import SmallWebRTCConnection
load_dotenv(override=True)
class TranscriptionLogger(FrameProcessor):
async def process_frame(self, frame: Frame, direction: FrameDirection):
await super().process_frame(frame, direction)
if isinstance(frame, TranscriptionFrame):
print(f"Transcription ({frame.language}): {frame.text}")
elif isinstance(frame, TranslationFrame):
print(f"Translation ({frame.language}): {frame.text}")
async def run_bot(webrtc_connection: SmallWebRTCConnection):
logger.info(f"Starting bot")
transport = SmallWebRTCTransport(
webrtc_connection=webrtc_connection,
params=TransportParams(audio_in_enabled=True),
)
stt = GladiaSTTService(
api_key=os.getenv("GLADIA_API_KEY"),
params=GladiaInputParams(
language_config=LanguageConfig(
languages=[Language.EN], # Input in English
code_switching=False,
),
realtime_processing=RealtimeProcessingConfig(
translation=True, # Enable translation
translation_config=TranslationConfig(
target_languages=[Language.ES], # Translate to Spanish
model="enhanced", # Use the enhanced translation model
),
),
),
)
tl = TranscriptionLogger()
pipeline = Pipeline([transport.input(), stt, tl])
task = PipelineTask(pipeline)
@transport.event_handler("on_client_disconnected")
async def on_client_disconnected(transport, client):
logger.info(f"Client disconnected")
@transport.event_handler("on_client_closed")
async def on_client_closed(transport, client):
logger.info(f"Client closed connection")
await task.cancel()
runner = PipelineRunner(handle_sigint=False)
await runner.run(task)
if __name__ == "__main__":
from run import main
main()

View File

@@ -256,6 +256,22 @@ class InterimTranscriptionFrame(TextFrame):
return f"{self.name}(user: {self.user_id}, text: [{self.text}], language: {self.language}, timestamp: {self.timestamp})" return f"{self.name}(user: {self.user_id}, text: [{self.text}], language: {self.language}, timestamp: {self.timestamp})"
@dataclass
class TranslationFrame(TextFrame):
"""A text frame with translated transcription data.
Will be placed in the transport's receive queue when a participant speaks.
"""
user_id: str
timestamp: str
language: Optional[Language] = None
def __str__(self):
return f"{self.name}(user: {self.user_id}, text: [{self.text}], language: {self.language}, timestamp: {self.timestamp})"
@dataclass @dataclass
class OpenAILLMContextAssistantTimestampFrame(DataFrame): class OpenAILLMContextAssistantTimestampFrame(DataFrame):
"""Timestamp information for assistant message in LLM context.""" """Timestamp information for assistant message in LLM context."""

View File

@@ -20,6 +20,7 @@ from pipecat.frames.frames import (
InterimTranscriptionFrame, InterimTranscriptionFrame,
StartFrame, StartFrame,
TranscriptionFrame, TranscriptionFrame,
TranslationFrame,
) )
from pipecat.services.gladia.config import GladiaInputParams from pipecat.services.gladia.config import GladiaInputParams
from pipecat.services.stt_service import STTService from pipecat.services.stt_service import STTService
@@ -384,16 +385,31 @@ class GladiaSTTService(STTService):
if content["type"] == "transcript": if content["type"] == "transcript":
utterance = content["data"]["utterance"] utterance = content["data"]["utterance"]
confidence = utterance.get("confidence", 0) confidence = utterance.get("confidence", 0)
language = utterance["language"]
transcript = utterance["text"] transcript = utterance["text"]
if confidence >= self._confidence: if confidence >= self._confidence:
if content["data"]["is_final"]: if content["data"]["is_final"]:
await self.push_frame( await self.push_frame(
TranscriptionFrame(transcript, "", time_now_iso8601()) TranscriptionFrame(transcript, "", time_now_iso8601(), language)
) )
else: else:
await self.push_frame( await self.push_frame(
InterimTranscriptionFrame(transcript, "", time_now_iso8601()) InterimTranscriptionFrame(
transcript, "", time_now_iso8601(), language
)
) )
elif content["type"] == "translation":
translated_utterance = content["data"]["translated_utterance"]
original_language = content["data"]["original_language"]
translated_language = translated_utterance["language"]
confidence = translated_utterance.get("confidence", 0)
translation = translated_utterance["text"]
if translated_language != original_language and confidence >= self._confidence:
await self.push_frame(
TranslationFrame(
translation, "", time_now_iso8601(), translated_language
)
)
except websockets.exceptions.ConnectionClosed: except websockets.exceptions.ConnectionClosed:
# Expected when closing the connection # Expected when closing the connection
pass pass