Merge pull request #1593 from WebinarGeek/wg/gladia-translations
Push gladia translations as a TranscriptionFrame
This commit is contained in:
@@ -9,6 +9,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
### Added
|
### Added
|
||||||
|
|
||||||
|
- Added `TranslationFrame`, a new frame type that contains a translated
|
||||||
|
transcription.
|
||||||
|
|
||||||
- Added `TransportParams.audio_in_passthrough`. If set (the default), incoming
|
- Added `TransportParams.audio_in_passthrough`. If set (the default), incoming
|
||||||
audio will be pushed downstream.
|
audio will be pushed downstream.
|
||||||
|
|
||||||
@@ -17,6 +20,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
### Changed
|
### Changed
|
||||||
|
|
||||||
|
- Updated `GladiaSTTService` to output a `TranslationFrame` when specifying a
|
||||||
|
`translation` and `translation_config`.
|
||||||
|
|
||||||
- STT services now passthrough audio frames by default. This allows you to add
|
- STT services now passthrough audio frames by default. This allows you to add
|
||||||
audio recording without worrying about what's wrong in your pipeline when it
|
audio recording without worrying about what's wrong in your pipeline when it
|
||||||
doesn't work the first time.
|
doesn't work the first time.
|
||||||
@@ -49,6 +55,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
- Added 04 foundational examples for client/server transports. Also, renamed
|
- Added 04 foundational examples for client/server transports. Also, renamed
|
||||||
`29-livekit-audio-chat.py` to `04b-transports-livekit.py`.
|
`29-livekit-audio-chat.py` to `04b-transports-livekit.py`.
|
||||||
|
|
||||||
|
- Added foundational example `13c-gladia-translation.py` showing how to use
|
||||||
|
`TranscriptionFrame` and `TranslationFrame`.
|
||||||
|
|
||||||
## [0.0.65] - 2025-04-23 "Sant Jordi's release" 🌹📕
|
## [0.0.65] - 2025-04-23 "Sant Jordi's release" 🌹📕
|
||||||
|
|
||||||
https://en.wikipedia.org/wiki/Saint_George%27s_Day_in_Catalonia
|
https://en.wikipedia.org/wiki/Saint_George%27s_Day_in_Catalonia
|
||||||
|
|||||||
@@ -130,6 +130,12 @@ pip install "pipecat-ai[option,...]"
|
|||||||
|
|
||||||
### Running tests
|
### Running tests
|
||||||
|
|
||||||
|
Install the test dependencies:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
pip install -r test-requirements.txt
|
||||||
|
```
|
||||||
|
|
||||||
From the root directory, run:
|
From the root directory, run:
|
||||||
|
|
||||||
```shell
|
```shell
|
||||||
|
|||||||
90
examples/foundational/13c-gladia-translation.py
Normal file
90
examples/foundational/13c-gladia-translation.py
Normal file
@@ -0,0 +1,90 @@
|
|||||||
|
#
|
||||||
|
# Copyright (c) 2024–2025, Daily
|
||||||
|
#
|
||||||
|
# SPDX-License-Identifier: BSD 2-Clause License
|
||||||
|
#
|
||||||
|
|
||||||
|
import os
|
||||||
|
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
from loguru import logger
|
||||||
|
|
||||||
|
from pipecat.frames.frames import Frame, TranscriptionFrame, TranslationFrame
|
||||||
|
from pipecat.pipeline.pipeline import Pipeline
|
||||||
|
from pipecat.pipeline.runner import PipelineRunner
|
||||||
|
from pipecat.pipeline.task import PipelineTask
|
||||||
|
from pipecat.processors.frame_processor import FrameDirection, FrameProcessor
|
||||||
|
from pipecat.services.gladia.config import (
|
||||||
|
GladiaInputParams,
|
||||||
|
LanguageConfig,
|
||||||
|
RealtimeProcessingConfig,
|
||||||
|
TranslationConfig,
|
||||||
|
)
|
||||||
|
from pipecat.services.gladia.stt import GladiaSTTService
|
||||||
|
from pipecat.transcriptions.language import Language
|
||||||
|
from pipecat.transports.base_transport import TransportParams
|
||||||
|
from pipecat.transports.network.small_webrtc import SmallWebRTCTransport
|
||||||
|
from pipecat.transports.network.webrtc_connection import SmallWebRTCConnection
|
||||||
|
|
||||||
|
load_dotenv(override=True)
|
||||||
|
|
||||||
|
|
||||||
|
class TranscriptionLogger(FrameProcessor):
|
||||||
|
async def process_frame(self, frame: Frame, direction: FrameDirection):
|
||||||
|
await super().process_frame(frame, direction)
|
||||||
|
|
||||||
|
if isinstance(frame, TranscriptionFrame):
|
||||||
|
print(f"Transcription ({frame.language}): {frame.text}")
|
||||||
|
elif isinstance(frame, TranslationFrame):
|
||||||
|
print(f"Translation ({frame.language}): {frame.text}")
|
||||||
|
|
||||||
|
|
||||||
|
async def run_bot(webrtc_connection: SmallWebRTCConnection):
|
||||||
|
logger.info(f"Starting bot")
|
||||||
|
|
||||||
|
transport = SmallWebRTCTransport(
|
||||||
|
webrtc_connection=webrtc_connection,
|
||||||
|
params=TransportParams(audio_in_enabled=True),
|
||||||
|
)
|
||||||
|
|
||||||
|
stt = GladiaSTTService(
|
||||||
|
api_key=os.getenv("GLADIA_API_KEY"),
|
||||||
|
params=GladiaInputParams(
|
||||||
|
language_config=LanguageConfig(
|
||||||
|
languages=[Language.EN], # Input in English
|
||||||
|
code_switching=False,
|
||||||
|
),
|
||||||
|
realtime_processing=RealtimeProcessingConfig(
|
||||||
|
translation=True, # Enable translation
|
||||||
|
translation_config=TranslationConfig(
|
||||||
|
target_languages=[Language.ES], # Translate to Spanish
|
||||||
|
model="enhanced", # Use the enhanced translation model
|
||||||
|
),
|
||||||
|
),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
tl = TranscriptionLogger()
|
||||||
|
|
||||||
|
pipeline = Pipeline([transport.input(), stt, tl])
|
||||||
|
|
||||||
|
task = PipelineTask(pipeline)
|
||||||
|
|
||||||
|
@transport.event_handler("on_client_disconnected")
|
||||||
|
async def on_client_disconnected(transport, client):
|
||||||
|
logger.info(f"Client disconnected")
|
||||||
|
|
||||||
|
@transport.event_handler("on_client_closed")
|
||||||
|
async def on_client_closed(transport, client):
|
||||||
|
logger.info(f"Client closed connection")
|
||||||
|
await task.cancel()
|
||||||
|
|
||||||
|
runner = PipelineRunner(handle_sigint=False)
|
||||||
|
|
||||||
|
await runner.run(task)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
from run import main
|
||||||
|
|
||||||
|
main()
|
||||||
@@ -256,6 +256,22 @@ class InterimTranscriptionFrame(TextFrame):
|
|||||||
return f"{self.name}(user: {self.user_id}, text: [{self.text}], language: {self.language}, timestamp: {self.timestamp})"
|
return f"{self.name}(user: {self.user_id}, text: [{self.text}], language: {self.language}, timestamp: {self.timestamp})"
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class TranslationFrame(TextFrame):
|
||||||
|
"""A text frame with translated transcription data.
|
||||||
|
|
||||||
|
Will be placed in the transport's receive queue when a participant speaks.
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
user_id: str
|
||||||
|
timestamp: str
|
||||||
|
language: Optional[Language] = None
|
||||||
|
|
||||||
|
def __str__(self):
|
||||||
|
return f"{self.name}(user: {self.user_id}, text: [{self.text}], language: {self.language}, timestamp: {self.timestamp})"
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class OpenAILLMContextAssistantTimestampFrame(DataFrame):
|
class OpenAILLMContextAssistantTimestampFrame(DataFrame):
|
||||||
"""Timestamp information for assistant message in LLM context."""
|
"""Timestamp information for assistant message in LLM context."""
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ from pipecat.frames.frames import (
|
|||||||
InterimTranscriptionFrame,
|
InterimTranscriptionFrame,
|
||||||
StartFrame,
|
StartFrame,
|
||||||
TranscriptionFrame,
|
TranscriptionFrame,
|
||||||
|
TranslationFrame,
|
||||||
)
|
)
|
||||||
from pipecat.services.gladia.config import GladiaInputParams
|
from pipecat.services.gladia.config import GladiaInputParams
|
||||||
from pipecat.services.stt_service import STTService
|
from pipecat.services.stt_service import STTService
|
||||||
@@ -384,16 +385,31 @@ class GladiaSTTService(STTService):
|
|||||||
if content["type"] == "transcript":
|
if content["type"] == "transcript":
|
||||||
utterance = content["data"]["utterance"]
|
utterance = content["data"]["utterance"]
|
||||||
confidence = utterance.get("confidence", 0)
|
confidence = utterance.get("confidence", 0)
|
||||||
|
language = utterance["language"]
|
||||||
transcript = utterance["text"]
|
transcript = utterance["text"]
|
||||||
if confidence >= self._confidence:
|
if confidence >= self._confidence:
|
||||||
if content["data"]["is_final"]:
|
if content["data"]["is_final"]:
|
||||||
await self.push_frame(
|
await self.push_frame(
|
||||||
TranscriptionFrame(transcript, "", time_now_iso8601())
|
TranscriptionFrame(transcript, "", time_now_iso8601(), language)
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
await self.push_frame(
|
await self.push_frame(
|
||||||
InterimTranscriptionFrame(transcript, "", time_now_iso8601())
|
InterimTranscriptionFrame(
|
||||||
|
transcript, "", time_now_iso8601(), language
|
||||||
|
)
|
||||||
)
|
)
|
||||||
|
elif content["type"] == "translation":
|
||||||
|
translated_utterance = content["data"]["translated_utterance"]
|
||||||
|
original_language = content["data"]["original_language"]
|
||||||
|
translated_language = translated_utterance["language"]
|
||||||
|
confidence = translated_utterance.get("confidence", 0)
|
||||||
|
translation = translated_utterance["text"]
|
||||||
|
if translated_language != original_language and confidence >= self._confidence:
|
||||||
|
await self.push_frame(
|
||||||
|
TranslationFrame(
|
||||||
|
translation, "", time_now_iso8601(), translated_language
|
||||||
|
)
|
||||||
|
)
|
||||||
except websockets.exceptions.ConnectionClosed:
|
except websockets.exceptions.ConnectionClosed:
|
||||||
# Expected when closing the connection
|
# Expected when closing the connection
|
||||||
pass
|
pass
|
||||||
|
|||||||
Reference in New Issue
Block a user