Add self._settings to 6 remaining services
- AWSNovaSonicLLMService: new `AWSNovaSonicLLMSettings` with `voice_id` and `endpointing_sensitivity`; remove `self._params` entirely, storing audio I/O config as plain instance variables - NeuphonicHttpTTSService: reuse `NeuphonicTTSSettings`; use inherited `language` field instead of bespoke `lang_code` - NvidiaTTSService: new `NvidiaTTSSettings` with `quality` - PiperTTSService / PiperHttpTTSService: new `PiperTTSSettings` / `PiperHttpTTSSettings` (no extra fields) - SpeechmaticsTTSService: new `SpeechmaticsTTSSettings` with `max_retries` Also remove redundant `lang_code` from `NeuphonicTTSSettings` (both WS and HTTP services now use the inherited `TTSSettings.language` field, with automatic enum conversion via the base class). HTTP services (Neuphonic HTTP, Piper HTTP, Speechmatics) don't override `_update_settings` since the base class applies changes to `self._settings` and subsequent requests read from it automatically.
This commit is contained in:
127
examples/foundational/55za-update-settings-neuphonic-http-tts.py
Normal file
127
examples/foundational/55za-update-settings-neuphonic-http-tts.py
Normal file
@@ -0,0 +1,127 @@
|
|||||||
|
#
|
||||||
|
# Copyright (c) 2024-2026, Daily
|
||||||
|
#
|
||||||
|
# SPDX-License-Identifier: BSD 2-Clause License
|
||||||
|
#
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import os
|
||||||
|
|
||||||
|
import aiohttp
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
from loguru import logger
|
||||||
|
|
||||||
|
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
||||||
|
from pipecat.frames.frames import LLMRunFrame, TTSUpdateSettingsFrame
|
||||||
|
from pipecat.pipeline.pipeline import Pipeline
|
||||||
|
from pipecat.pipeline.runner import PipelineRunner
|
||||||
|
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
||||||
|
from pipecat.processors.aggregators.llm_context import LLMContext
|
||||||
|
from pipecat.processors.aggregators.llm_response_universal import (
|
||||||
|
LLMContextAggregatorPair,
|
||||||
|
LLMUserAggregatorParams,
|
||||||
|
)
|
||||||
|
from pipecat.runner.types import RunnerArguments
|
||||||
|
from pipecat.runner.utils import create_transport
|
||||||
|
from pipecat.services.deepgram.stt import DeepgramSTTService
|
||||||
|
from pipecat.services.neuphonic.tts import NeuphonicHttpTTSService, NeuphonicTTSSettings
|
||||||
|
from pipecat.services.openai.llm import OpenAILLMService
|
||||||
|
from pipecat.transports.base_transport import BaseTransport, TransportParams
|
||||||
|
from pipecat.transports.daily.transport import DailyParams
|
||||||
|
from pipecat.transports.websocket.fastapi import FastAPIWebsocketParams
|
||||||
|
|
||||||
|
load_dotenv(override=True)
|
||||||
|
|
||||||
|
transport_params = {
|
||||||
|
"daily": lambda: DailyParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
"twilio": lambda: FastAPIWebsocketParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
"webrtc": lambda: TransportParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def run_bot(transport: BaseTransport, runner_args: RunnerArguments):
|
||||||
|
logger.info(f"Starting bot")
|
||||||
|
|
||||||
|
stt = DeepgramSTTService(api_key=os.getenv("DEEPGRAM_API_KEY"))
|
||||||
|
|
||||||
|
async with aiohttp.ClientSession() as session:
|
||||||
|
tts = NeuphonicHttpTTSService(
|
||||||
|
api_key=os.getenv("NEUPHONIC_API_KEY"),
|
||||||
|
aiohttp_session=session,
|
||||||
|
)
|
||||||
|
|
||||||
|
llm = OpenAILLMService(api_key=os.getenv("OPENAI_API_KEY"))
|
||||||
|
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "system",
|
||||||
|
"content": "You are a helpful LLM in a WebRTC call. Your goal is to demonstrate your capabilities in a succinct way. Your output will be spoken aloud, so avoid special characters that can't easily be spoken, such as emojis or bullet points. Respond to what the user said in a creative and helpful way.",
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
context = LLMContext(messages)
|
||||||
|
user_aggregator, assistant_aggregator = LLMContextAggregatorPair(
|
||||||
|
context,
|
||||||
|
user_params=LLMUserAggregatorParams(vad_analyzer=SileroVADAnalyzer()),
|
||||||
|
)
|
||||||
|
|
||||||
|
pipeline = Pipeline(
|
||||||
|
[
|
||||||
|
transport.input(),
|
||||||
|
stt,
|
||||||
|
user_aggregator,
|
||||||
|
llm,
|
||||||
|
tts,
|
||||||
|
transport.output(),
|
||||||
|
assistant_aggregator,
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
task = PipelineTask(
|
||||||
|
pipeline,
|
||||||
|
params=PipelineParams(
|
||||||
|
enable_metrics=True,
|
||||||
|
enable_usage_metrics=True,
|
||||||
|
),
|
||||||
|
idle_timeout_secs=runner_args.pipeline_idle_timeout_secs,
|
||||||
|
)
|
||||||
|
|
||||||
|
@transport.event_handler("on_client_connected")
|
||||||
|
async def on_client_connected(transport, client):
|
||||||
|
logger.info(f"Client connected")
|
||||||
|
messages.append({"role": "system", "content": "Please introduce yourself to the user."})
|
||||||
|
await task.queue_frames([LLMRunFrame()])
|
||||||
|
|
||||||
|
await asyncio.sleep(10)
|
||||||
|
logger.info("Updating Neuphonic HTTP TTS settings: speed=1.4")
|
||||||
|
await task.queue_frame(TTSUpdateSettingsFrame(update=NeuphonicTTSSettings(speed=1.4)))
|
||||||
|
|
||||||
|
@transport.event_handler("on_client_disconnected")
|
||||||
|
async def on_client_disconnected(transport, client):
|
||||||
|
logger.info(f"Client disconnected")
|
||||||
|
await task.cancel()
|
||||||
|
|
||||||
|
runner = PipelineRunner(handle_sigint=runner_args.handle_sigint)
|
||||||
|
|
||||||
|
await runner.run(task)
|
||||||
|
|
||||||
|
|
||||||
|
async def bot(runner_args: RunnerArguments):
|
||||||
|
"""Main bot entry point compatible with Pipecat Cloud."""
|
||||||
|
transport = await create_transport(runner_args, transport_params)
|
||||||
|
await run_bot(transport, runner_args)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
from pipecat.runner.run import main
|
||||||
|
|
||||||
|
main()
|
||||||
@@ -0,0 +1,124 @@
|
|||||||
|
#
|
||||||
|
# Copyright (c) 2024-2026, Daily
|
||||||
|
#
|
||||||
|
# SPDX-License-Identifier: BSD 2-Clause License
|
||||||
|
#
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import os
|
||||||
|
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
from loguru import logger
|
||||||
|
|
||||||
|
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
||||||
|
from pipecat.frames.frames import LLMRunFrame, LLMUpdateSettingsFrame
|
||||||
|
from pipecat.pipeline.pipeline import Pipeline
|
||||||
|
from pipecat.pipeline.runner import PipelineRunner
|
||||||
|
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
||||||
|
from pipecat.processors.aggregators.llm_context import LLMContext
|
||||||
|
from pipecat.processors.aggregators.llm_response_universal import (
|
||||||
|
LLMContextAggregatorPair,
|
||||||
|
LLMUserAggregatorParams,
|
||||||
|
)
|
||||||
|
from pipecat.runner.types import RunnerArguments
|
||||||
|
from pipecat.runner.utils import create_transport
|
||||||
|
from pipecat.services.aws.nova_sonic.llm import AWSNovaSonicLLMService, AWSNovaSonicLLMSettings
|
||||||
|
from pipecat.transports.base_transport import BaseTransport, TransportParams
|
||||||
|
from pipecat.transports.daily.transport import DailyParams
|
||||||
|
from pipecat.transports.websocket.fastapi import FastAPIWebsocketParams
|
||||||
|
|
||||||
|
load_dotenv(override=True)
|
||||||
|
|
||||||
|
transport_params = {
|
||||||
|
"daily": lambda: DailyParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
"twilio": lambda: FastAPIWebsocketParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
"webrtc": lambda: TransportParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def run_bot(transport: BaseTransport, runner_args: RunnerArguments):
|
||||||
|
logger.info(f"Starting bot")
|
||||||
|
|
||||||
|
llm = AWSNovaSonicLLMService(
|
||||||
|
secret_access_key=os.getenv("AWS_SECRET_ACCESS_KEY"),
|
||||||
|
access_key_id=os.getenv("AWS_ACCESS_KEY_ID"),
|
||||||
|
region=os.getenv("AWS_REGION"),
|
||||||
|
)
|
||||||
|
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "system",
|
||||||
|
"content": "You are a helpful LLM in a WebRTC call. Your goal is to demonstrate your capabilities in a succinct way. Your output will be spoken aloud, so avoid special characters that can't easily be spoken, such as emojis or bullet points. Respond to what the user said in a creative and helpful way.",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Tell me a fun fact!",
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
context = LLMContext(messages)
|
||||||
|
user_aggregator, assistant_aggregator = LLMContextAggregatorPair(
|
||||||
|
context,
|
||||||
|
user_params=LLMUserAggregatorParams(vad_analyzer=SileroVADAnalyzer()),
|
||||||
|
)
|
||||||
|
|
||||||
|
pipeline = Pipeline(
|
||||||
|
[
|
||||||
|
transport.input(),
|
||||||
|
user_aggregator,
|
||||||
|
llm,
|
||||||
|
transport.output(),
|
||||||
|
assistant_aggregator,
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
task = PipelineTask(
|
||||||
|
pipeline,
|
||||||
|
params=PipelineParams(
|
||||||
|
enable_metrics=True,
|
||||||
|
enable_usage_metrics=True,
|
||||||
|
),
|
||||||
|
idle_timeout_secs=runner_args.pipeline_idle_timeout_secs,
|
||||||
|
)
|
||||||
|
|
||||||
|
@transport.event_handler("on_client_connected")
|
||||||
|
async def on_client_connected(transport, client):
|
||||||
|
logger.info(f"Client connected")
|
||||||
|
messages.append({"role": "system", "content": "Please introduce yourself to the user."})
|
||||||
|
await task.queue_frames([LLMRunFrame()])
|
||||||
|
|
||||||
|
await asyncio.sleep(10)
|
||||||
|
logger.info("Updating AWS Nova Sonic LLM settings: temperature=0.1")
|
||||||
|
await task.queue_frame(
|
||||||
|
LLMUpdateSettingsFrame(update=AWSNovaSonicLLMSettings(temperature=0.1))
|
||||||
|
)
|
||||||
|
|
||||||
|
@transport.event_handler("on_client_disconnected")
|
||||||
|
async def on_client_disconnected(transport, client):
|
||||||
|
logger.info(f"Client disconnected")
|
||||||
|
await task.cancel()
|
||||||
|
|
||||||
|
runner = PipelineRunner(handle_sigint=runner_args.handle_sigint)
|
||||||
|
|
||||||
|
await runner.run(task)
|
||||||
|
|
||||||
|
|
||||||
|
async def bot(runner_args: RunnerArguments):
|
||||||
|
"""Main bot entry point compatible with Pipecat Cloud."""
|
||||||
|
transport = await create_transport(runner_args, transport_params)
|
||||||
|
await run_bot(transport, runner_args)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
from pipecat.runner.run import main
|
||||||
|
|
||||||
|
main()
|
||||||
125
examples/foundational/55zzl-update-settings-nvidia-tts.py
Normal file
125
examples/foundational/55zzl-update-settings-nvidia-tts.py
Normal file
@@ -0,0 +1,125 @@
|
|||||||
|
#
|
||||||
|
# Copyright (c) 2024-2026, Daily
|
||||||
|
#
|
||||||
|
# SPDX-License-Identifier: BSD 2-Clause License
|
||||||
|
#
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import os
|
||||||
|
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
from loguru import logger
|
||||||
|
|
||||||
|
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
||||||
|
from pipecat.frames.frames import LLMRunFrame, TTSUpdateSettingsFrame
|
||||||
|
from pipecat.pipeline.pipeline import Pipeline
|
||||||
|
from pipecat.pipeline.runner import PipelineRunner
|
||||||
|
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
||||||
|
from pipecat.processors.aggregators.llm_context import LLMContext
|
||||||
|
from pipecat.processors.aggregators.llm_response_universal import (
|
||||||
|
LLMContextAggregatorPair,
|
||||||
|
LLMUserAggregatorParams,
|
||||||
|
)
|
||||||
|
from pipecat.runner.types import RunnerArguments
|
||||||
|
from pipecat.runner.utils import create_transport
|
||||||
|
from pipecat.services.deepgram.stt import DeepgramSTTService
|
||||||
|
from pipecat.services.nvidia.tts import NvidiaTTSService, NvidiaTTSSettings
|
||||||
|
from pipecat.services.openai.llm import OpenAILLMService
|
||||||
|
from pipecat.transcriptions.language import Language
|
||||||
|
from pipecat.transports.base_transport import BaseTransport, TransportParams
|
||||||
|
from pipecat.transports.daily.transport import DailyParams
|
||||||
|
from pipecat.transports.websocket.fastapi import FastAPIWebsocketParams
|
||||||
|
|
||||||
|
load_dotenv(override=True)
|
||||||
|
|
||||||
|
transport_params = {
|
||||||
|
"daily": lambda: DailyParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
"twilio": lambda: FastAPIWebsocketParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
"webrtc": lambda: TransportParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def run_bot(transport: BaseTransport, runner_args: RunnerArguments):
|
||||||
|
logger.info(f"Starting bot")
|
||||||
|
|
||||||
|
stt = DeepgramSTTService(api_key=os.getenv("DEEPGRAM_API_KEY"))
|
||||||
|
|
||||||
|
tts = NvidiaTTSService(api_key=os.getenv("NVIDIA_API_KEY"))
|
||||||
|
|
||||||
|
llm = OpenAILLMService(api_key=os.getenv("OPENAI_API_KEY"))
|
||||||
|
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "system",
|
||||||
|
"content": "You are a helpful LLM in a WebRTC call. Your goal is to demonstrate your capabilities in a succinct way. Your output will be spoken aloud, so avoid special characters that can't easily be spoken, such as emojis or bullet points. Respond to what the user said in a creative and helpful way.",
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
context = LLMContext(messages)
|
||||||
|
user_aggregator, assistant_aggregator = LLMContextAggregatorPair(
|
||||||
|
context,
|
||||||
|
user_params=LLMUserAggregatorParams(vad_analyzer=SileroVADAnalyzer()),
|
||||||
|
)
|
||||||
|
|
||||||
|
pipeline = Pipeline(
|
||||||
|
[
|
||||||
|
transport.input(),
|
||||||
|
stt,
|
||||||
|
user_aggregator,
|
||||||
|
llm,
|
||||||
|
tts,
|
||||||
|
transport.output(),
|
||||||
|
assistant_aggregator,
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
task = PipelineTask(
|
||||||
|
pipeline,
|
||||||
|
params=PipelineParams(
|
||||||
|
enable_metrics=True,
|
||||||
|
enable_usage_metrics=True,
|
||||||
|
),
|
||||||
|
idle_timeout_secs=runner_args.pipeline_idle_timeout_secs,
|
||||||
|
)
|
||||||
|
|
||||||
|
@transport.event_handler("on_client_connected")
|
||||||
|
async def on_client_connected(transport, client):
|
||||||
|
logger.info(f"Client connected")
|
||||||
|
messages.append({"role": "system", "content": "Please introduce yourself to the user."})
|
||||||
|
await task.queue_frames([LLMRunFrame()])
|
||||||
|
|
||||||
|
await asyncio.sleep(10)
|
||||||
|
logger.info('Updating NVIDIA TTS settings: language="ES_US"')
|
||||||
|
await task.queue_frame(
|
||||||
|
TTSUpdateSettingsFrame(update=NvidiaTTSSettings(language=Language.ES_US))
|
||||||
|
)
|
||||||
|
|
||||||
|
@transport.event_handler("on_client_disconnected")
|
||||||
|
async def on_client_disconnected(transport, client):
|
||||||
|
logger.info(f"Client disconnected")
|
||||||
|
await task.cancel()
|
||||||
|
|
||||||
|
runner = PipelineRunner(handle_sigint=runner_args.handle_sigint)
|
||||||
|
|
||||||
|
await runner.run(task)
|
||||||
|
|
||||||
|
|
||||||
|
async def bot(runner_args: RunnerArguments):
|
||||||
|
"""Main bot entry point compatible with Pipecat Cloud."""
|
||||||
|
transport = await create_transport(runner_args, transport_params)
|
||||||
|
await run_bot(transport, runner_args)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
from pipecat.runner.run import main
|
||||||
|
|
||||||
|
main()
|
||||||
129
examples/foundational/55zzm-update-settings-speechmatics-tts.py
Normal file
129
examples/foundational/55zzm-update-settings-speechmatics-tts.py
Normal file
@@ -0,0 +1,129 @@
|
|||||||
|
#
|
||||||
|
# Copyright (c) 2024-2026, Daily
|
||||||
|
#
|
||||||
|
# SPDX-License-Identifier: BSD 2-Clause License
|
||||||
|
#
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import os
|
||||||
|
|
||||||
|
import aiohttp
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
from loguru import logger
|
||||||
|
|
||||||
|
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
||||||
|
from pipecat.frames.frames import LLMRunFrame, TTSUpdateSettingsFrame
|
||||||
|
from pipecat.pipeline.pipeline import Pipeline
|
||||||
|
from pipecat.pipeline.runner import PipelineRunner
|
||||||
|
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
||||||
|
from pipecat.processors.aggregators.llm_context import LLMContext
|
||||||
|
from pipecat.processors.aggregators.llm_response_universal import (
|
||||||
|
LLMContextAggregatorPair,
|
||||||
|
LLMUserAggregatorParams,
|
||||||
|
)
|
||||||
|
from pipecat.runner.types import RunnerArguments
|
||||||
|
from pipecat.runner.utils import create_transport
|
||||||
|
from pipecat.services.deepgram.stt import DeepgramSTTService
|
||||||
|
from pipecat.services.openai.llm import OpenAILLMService
|
||||||
|
from pipecat.services.speechmatics.tts import SpeechmaticsTTSService, SpeechmaticsTTSSettings
|
||||||
|
from pipecat.transports.base_transport import BaseTransport, TransportParams
|
||||||
|
from pipecat.transports.daily.transport import DailyParams
|
||||||
|
from pipecat.transports.websocket.fastapi import FastAPIWebsocketParams
|
||||||
|
|
||||||
|
load_dotenv(override=True)
|
||||||
|
|
||||||
|
transport_params = {
|
||||||
|
"daily": lambda: DailyParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
"twilio": lambda: FastAPIWebsocketParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
"webrtc": lambda: TransportParams(
|
||||||
|
audio_in_enabled=True,
|
||||||
|
audio_out_enabled=True,
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def run_bot(transport: BaseTransport, runner_args: RunnerArguments):
|
||||||
|
logger.info(f"Starting bot")
|
||||||
|
|
||||||
|
stt = DeepgramSTTService(api_key=os.getenv("DEEPGRAM_API_KEY"))
|
||||||
|
|
||||||
|
async with aiohttp.ClientSession() as session:
|
||||||
|
tts = SpeechmaticsTTSService(
|
||||||
|
api_key=os.getenv("SPEECHMATICS_API_KEY"),
|
||||||
|
aiohttp_session=session,
|
||||||
|
)
|
||||||
|
|
||||||
|
llm = OpenAILLMService(api_key=os.getenv("OPENAI_API_KEY"))
|
||||||
|
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "system",
|
||||||
|
"content": "You are a helpful LLM in a WebRTC call. Your goal is to demonstrate your capabilities in a succinct way. Your output will be spoken aloud, so avoid special characters that can't easily be spoken, such as emojis or bullet points. Respond to what the user said in a creative and helpful way.",
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
context = LLMContext(messages)
|
||||||
|
user_aggregator, assistant_aggregator = LLMContextAggregatorPair(
|
||||||
|
context,
|
||||||
|
user_params=LLMUserAggregatorParams(vad_analyzer=SileroVADAnalyzer()),
|
||||||
|
)
|
||||||
|
|
||||||
|
pipeline = Pipeline(
|
||||||
|
[
|
||||||
|
transport.input(),
|
||||||
|
stt,
|
||||||
|
user_aggregator,
|
||||||
|
llm,
|
||||||
|
tts,
|
||||||
|
transport.output(),
|
||||||
|
assistant_aggregator,
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
task = PipelineTask(
|
||||||
|
pipeline,
|
||||||
|
params=PipelineParams(
|
||||||
|
enable_metrics=True,
|
||||||
|
enable_usage_metrics=True,
|
||||||
|
),
|
||||||
|
idle_timeout_secs=runner_args.pipeline_idle_timeout_secs,
|
||||||
|
)
|
||||||
|
|
||||||
|
@transport.event_handler("on_client_connected")
|
||||||
|
async def on_client_connected(transport, client):
|
||||||
|
logger.info(f"Client connected")
|
||||||
|
messages.append({"role": "system", "content": "Please introduce yourself to the user."})
|
||||||
|
await task.queue_frames([LLMRunFrame()])
|
||||||
|
|
||||||
|
await asyncio.sleep(10)
|
||||||
|
logger.info('Updating Speechmatics TTS settings: voice="theo"')
|
||||||
|
await task.queue_frame(
|
||||||
|
TTSUpdateSettingsFrame(update=SpeechmaticsTTSSettings(voice="theo"))
|
||||||
|
)
|
||||||
|
|
||||||
|
@transport.event_handler("on_client_disconnected")
|
||||||
|
async def on_client_disconnected(transport, client):
|
||||||
|
logger.info(f"Client disconnected")
|
||||||
|
await task.cancel()
|
||||||
|
|
||||||
|
runner = PipelineRunner(handle_sigint=runner_args.handle_sigint)
|
||||||
|
|
||||||
|
await runner.run(task)
|
||||||
|
|
||||||
|
|
||||||
|
async def bot(runner_args: RunnerArguments):
|
||||||
|
"""Main bot entry point compatible with Pipecat Cloud."""
|
||||||
|
transport = await create_transport(runner_args, transport_params)
|
||||||
|
await run_bot(transport, runner_args)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
from pipecat.runner.run import main
|
||||||
|
|
||||||
|
main()
|
||||||
@@ -16,7 +16,7 @@ import json
|
|||||||
import time
|
import time
|
||||||
import uuid
|
import uuid
|
||||||
import wave
|
import wave
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass, field
|
||||||
from enum import Enum
|
from enum import Enum
|
||||||
from importlib.resources import files
|
from importlib.resources import files
|
||||||
from typing import Any, List, Optional
|
from typing import Any, List, Optional
|
||||||
@@ -60,7 +60,7 @@ from pipecat.processors.aggregators.openai_llm_context import (
|
|||||||
)
|
)
|
||||||
from pipecat.processors.frame_processor import FrameDirection
|
from pipecat.processors.frame_processor import FrameDirection
|
||||||
from pipecat.services.llm_service import LLMService
|
from pipecat.services.llm_service import LLMService
|
||||||
from pipecat.services.settings import LLMSettings
|
from pipecat.services.settings import NOT_GIVEN, LLMSettings, _NotGiven
|
||||||
from pipecat.utils.time import time_now_iso8601
|
from pipecat.utils.time import time_now_iso8601
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@@ -186,6 +186,20 @@ class Params(BaseModel):
|
|||||||
endpointing_sensitivity: Optional[str] = Field(default=None)
|
endpointing_sensitivity: Optional[str] = Field(default=None)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class AWSNovaSonicLLMSettings(LLMSettings):
|
||||||
|
"""Settings for AWS Nova Sonic LLM service.
|
||||||
|
|
||||||
|
Parameters:
|
||||||
|
voice_id: Voice for speech synthesis.
|
||||||
|
endpointing_sensitivity: Controls how quickly Nova Sonic decides the
|
||||||
|
user has stopped speaking. Can be "LOW", "MEDIUM", or "HIGH".
|
||||||
|
"""
|
||||||
|
|
||||||
|
voice_id: str | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
||||||
|
endpointing_sensitivity: str | None | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
||||||
|
|
||||||
|
|
||||||
class AWSNovaSonicLLMService(LLMService):
|
class AWSNovaSonicLLMService(LLMService):
|
||||||
"""AWS Nova Sonic speech-to-speech LLM service.
|
"""AWS Nova Sonic speech-to-speech LLM service.
|
||||||
|
|
||||||
@@ -193,6 +207,8 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
and function calling capabilities using AWS Nova Sonic model.
|
and function calling capabilities using AWS Nova Sonic model.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
_settings: AWSNovaSonicLLMSettings
|
||||||
|
|
||||||
# Override the default adapter to use the AWSNovaSonicLLMAdapter one
|
# Override the default adapter to use the AWSNovaSonicLLMAdapter one
|
||||||
adapter_class = AWSNovaSonicLLMAdapter
|
adapter_class = AWSNovaSonicLLMAdapter
|
||||||
|
|
||||||
@@ -243,23 +259,38 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
self._access_key_id = access_key_id
|
self._access_key_id = access_key_id
|
||||||
self._session_token = session_token
|
self._session_token = session_token
|
||||||
self._region = region
|
self._region = region
|
||||||
self._model = model
|
|
||||||
self._client: Optional[BedrockRuntimeClient] = None
|
self._client: Optional[BedrockRuntimeClient] = None
|
||||||
self._voice_id = voice_id
|
params = params or Params()
|
||||||
self._params = params or Params()
|
self._settings = AWSNovaSonicLLMSettings(
|
||||||
|
model=model,
|
||||||
|
voice_id=voice_id,
|
||||||
|
temperature=params.temperature,
|
||||||
|
max_tokens=params.max_tokens,
|
||||||
|
top_p=params.top_p,
|
||||||
|
endpointing_sensitivity=params.endpointing_sensitivity,
|
||||||
|
)
|
||||||
|
self.set_model_name(model)
|
||||||
|
|
||||||
|
# Audio I/O config (hardware settings, not runtime-tunable)
|
||||||
|
self._input_sample_rate = params.input_sample_rate
|
||||||
|
self._input_sample_size = params.input_sample_size
|
||||||
|
self._input_channel_count = params.input_channel_count
|
||||||
|
self._output_sample_rate = params.output_sample_rate
|
||||||
|
self._output_sample_size = params.output_sample_size
|
||||||
|
self._output_channel_count = params.output_channel_count
|
||||||
self._system_instruction = system_instruction
|
self._system_instruction = system_instruction
|
||||||
self._tools = tools
|
self._tools = tools
|
||||||
|
|
||||||
# Validate endpointing_sensitivity parameter
|
# Validate endpointing_sensitivity parameter
|
||||||
if (
|
if (
|
||||||
self._params.endpointing_sensitivity
|
self._settings.endpointing_sensitivity
|
||||||
and not self._is_endpointing_sensitivity_supported()
|
and not self._is_endpointing_sensitivity_supported()
|
||||||
):
|
):
|
||||||
logger.warning(
|
logger.warning(
|
||||||
f"endpointing_sensitivity is not supported for model '{model}' and will be ignored. "
|
f"endpointing_sensitivity is not supported for model '{model}' and will be ignored. "
|
||||||
"This parameter is only supported starting with Nova 2 Sonic (amazon.nova-2-sonic-v1:0)."
|
"This parameter is only supported starting with Nova 2 Sonic (amazon.nova-2-sonic-v1:0)."
|
||||||
)
|
)
|
||||||
self._params.endpointing_sensitivity = None
|
self._settings.endpointing_sensitivity = None
|
||||||
|
|
||||||
if not send_transcription_frames:
|
if not send_transcription_frames:
|
||||||
import warnings
|
import warnings
|
||||||
@@ -307,7 +338,7 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
# settings
|
# settings
|
||||||
#
|
#
|
||||||
|
|
||||||
async def _update_settings(self, update: LLMSettings) -> dict[str, Any]:
|
async def _update_settings(self, update: AWSNovaSonicLLMSettings) -> dict[str, Any]:
|
||||||
"""Apply a settings update.
|
"""Apply a settings update.
|
||||||
|
|
||||||
Settings are stored but not applied to the active connection.
|
Settings are stored but not applied to the active connection.
|
||||||
@@ -320,7 +351,7 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
# TODO: someday we could reconnect here to apply updated settings.
|
# TODO: someday we could reconnect here to apply updated settings.
|
||||||
# Code might look something like the below:
|
# Code might look something like the below:
|
||||||
# await self._disconnect()
|
# await self._disconnect()
|
||||||
# await self._connect()
|
# await self._start_connecting()
|
||||||
|
|
||||||
self._warn_unhandled_updated_settings(changed)
|
self._warn_unhandled_updated_settings(changed)
|
||||||
|
|
||||||
@@ -496,7 +527,7 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
|
|
||||||
# Start the bidirectional stream
|
# Start the bidirectional stream
|
||||||
self._stream = await self._client.invoke_model_with_bidirectional_stream(
|
self._stream = await self._client.invoke_model_with_bidirectional_stream(
|
||||||
InvokeModelWithBidirectionalStreamOperationInput(model_id=self._model)
|
InvokeModelWithBidirectionalStreamOperationInput(model_id=self._settings.model)
|
||||||
)
|
)
|
||||||
|
|
||||||
# Send session start event
|
# Send session start event
|
||||||
@@ -663,7 +694,7 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
|
|
||||||
def _is_first_generation_sonic_model(self) -> bool:
|
def _is_first_generation_sonic_model(self) -> bool:
|
||||||
# Nova Sonic (the older model) is identified by "amazon.nova-sonic-v1:0"
|
# Nova Sonic (the older model) is identified by "amazon.nova-sonic-v1:0"
|
||||||
return self._model == "amazon.nova-sonic-v1:0"
|
return self._settings.model == "amazon.nova-sonic-v1:0"
|
||||||
|
|
||||||
def _is_endpointing_sensitivity_supported(self) -> bool:
|
def _is_endpointing_sensitivity_supported(self) -> bool:
|
||||||
# endpointing_sensitivity is only supported with Nova 2 Sonic (and,
|
# endpointing_sensitivity is only supported with Nova 2 Sonic (and,
|
||||||
@@ -682,9 +713,9 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
turn_detection_config = (
|
turn_detection_config = (
|
||||||
f""",
|
f""",
|
||||||
"turnDetectionConfiguration": {{
|
"turnDetectionConfiguration": {{
|
||||||
"endpointingSensitivity": "{self._params.endpointing_sensitivity}"
|
"endpointingSensitivity": "{self._settings.endpointing_sensitivity}"
|
||||||
}}"""
|
}}"""
|
||||||
if self._params.endpointing_sensitivity
|
if self._settings.endpointing_sensitivity
|
||||||
else ""
|
else ""
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -693,9 +724,9 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
"event": {{
|
"event": {{
|
||||||
"sessionStart": {{
|
"sessionStart": {{
|
||||||
"inferenceConfiguration": {{
|
"inferenceConfiguration": {{
|
||||||
"maxTokens": {self._params.max_tokens},
|
"maxTokens": {self._settings.max_tokens},
|
||||||
"topP": {self._params.top_p},
|
"topP": {self._settings.top_p},
|
||||||
"temperature": {self._params.temperature}
|
"temperature": {self._settings.temperature}
|
||||||
}}{turn_detection_config}
|
}}{turn_detection_config}
|
||||||
}}
|
}}
|
||||||
}}
|
}}
|
||||||
@@ -730,10 +761,10 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
}},
|
}},
|
||||||
"audioOutputConfiguration": {{
|
"audioOutputConfiguration": {{
|
||||||
"mediaType": "audio/lpcm",
|
"mediaType": "audio/lpcm",
|
||||||
"sampleRateHertz": {self._params.output_sample_rate},
|
"sampleRateHertz": {self._output_sample_rate},
|
||||||
"sampleSizeBits": {self._params.output_sample_size},
|
"sampleSizeBits": {self._output_sample_size},
|
||||||
"channelCount": {self._params.output_channel_count},
|
"channelCount": {self._output_channel_count},
|
||||||
"voiceId": "{self._voice_id}",
|
"voiceId": "{self._settings.voice_id}",
|
||||||
"encoding": "base64",
|
"encoding": "base64",
|
||||||
"audioType": "SPEECH"
|
"audioType": "SPEECH"
|
||||||
}}{tools_config}
|
}}{tools_config}
|
||||||
@@ -758,9 +789,9 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
"role": "USER",
|
"role": "USER",
|
||||||
"audioInputConfiguration": {{
|
"audioInputConfiguration": {{
|
||||||
"mediaType": "audio/lpcm",
|
"mediaType": "audio/lpcm",
|
||||||
"sampleRateHertz": {self._params.input_sample_rate},
|
"sampleRateHertz": {self._input_sample_rate},
|
||||||
"sampleSizeBits": {self._params.input_sample_size},
|
"sampleSizeBits": {self._input_sample_size},
|
||||||
"channelCount": {self._params.input_channel_count},
|
"channelCount": {self._input_channel_count},
|
||||||
"audioType": "SPEECH",
|
"audioType": "SPEECH",
|
||||||
"encoding": "base64"
|
"encoding": "base64"
|
||||||
}}
|
}}
|
||||||
@@ -1043,8 +1074,8 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
audio = base64.b64decode(audio_content)
|
audio = base64.b64decode(audio_content)
|
||||||
frame = TTSAudioRawFrame(
|
frame = TTSAudioRawFrame(
|
||||||
audio=audio,
|
audio=audio,
|
||||||
sample_rate=self._params.output_sample_rate,
|
sample_rate=self._output_sample_rate,
|
||||||
num_channels=self._params.output_channel_count,
|
num_channels=self._output_channel_count,
|
||||||
)
|
)
|
||||||
await self.push_frame(frame)
|
await self.push_frame(frame)
|
||||||
|
|
||||||
@@ -1328,7 +1359,7 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
"""
|
"""
|
||||||
if not self._is_assistant_response_trigger_needed():
|
if not self._is_assistant_response_trigger_needed():
|
||||||
logger.warning(
|
logger.warning(
|
||||||
f"Assistant response trigger not needed for model '{self._model}'; skipping. "
|
f"Assistant response trigger not needed for model '{self._settings.model}'; skipping. "
|
||||||
"An LLMRunFrame() should be sufficient to prompt the assistant to respond, "
|
"An LLMRunFrame() should be sufficient to prompt the assistant to respond, "
|
||||||
"assuming the context ends in a user message."
|
"assuming the context ends in a user message."
|
||||||
)
|
)
|
||||||
@@ -1356,9 +1387,9 @@ class AWSNovaSonicLLMService(LLMService):
|
|||||||
chunk_duration = 0.02 # what we might get from InputAudioRawFrame
|
chunk_duration = 0.02 # what we might get from InputAudioRawFrame
|
||||||
chunk_size = int(
|
chunk_size = int(
|
||||||
chunk_duration
|
chunk_duration
|
||||||
* self._params.input_sample_rate
|
* self._input_sample_rate
|
||||||
* self._params.input_channel_count
|
* self._input_channel_count
|
||||||
* (self._params.input_sample_size / 8)
|
* (self._input_sample_size / 8)
|
||||||
) # e.g. 0.02 seconds of 16-bit (2-byte) PCM mono audio at 16kHz is 640 bytes
|
) # e.g. 0.02 seconds of 16-bit (2-byte) PCM mono audio at 16kHz is 640 bytes
|
||||||
|
|
||||||
# Lead with a bit of blank audio, if needed.
|
# Lead with a bit of blank audio, if needed.
|
||||||
|
|||||||
@@ -79,13 +79,11 @@ class NeuphonicTTSSettings(TTSSettings):
|
|||||||
"""Settings for Neuphonic TTS service.
|
"""Settings for Neuphonic TTS service.
|
||||||
|
|
||||||
Parameters:
|
Parameters:
|
||||||
lang_code: Neuphonic language code.
|
|
||||||
speed: Speech speed multiplier. Defaults to 1.0.
|
speed: Speech speed multiplier. Defaults to 1.0.
|
||||||
encoding: Audio encoding format.
|
encoding: Audio encoding format.
|
||||||
sampling_rate: Audio sample rate.
|
sampling_rate: Audio sample rate.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
lang_code: str | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
|
||||||
speed: float | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
speed: float | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
||||||
encoding: str | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
encoding: str | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
||||||
sampling_rate: int | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
sampling_rate: int | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
||||||
@@ -149,7 +147,7 @@ class NeuphonicTTSService(InterruptibleTTSService):
|
|||||||
self._api_key = api_key
|
self._api_key = api_key
|
||||||
self._url = url
|
self._url = url
|
||||||
self._settings = NeuphonicTTSSettings(
|
self._settings = NeuphonicTTSSettings(
|
||||||
lang_code=self.language_to_service_language(params.language),
|
language=self.language_to_service_language(params.language),
|
||||||
speed=params.speed,
|
speed=params.speed,
|
||||||
encoding=encoding,
|
encoding=encoding,
|
||||||
sampling_rate=sample_rate,
|
sampling_rate=sample_rate,
|
||||||
@@ -286,7 +284,7 @@ class NeuphonicTTSService(InterruptibleTTSService):
|
|||||||
logger.debug("Connecting to Neuphonic")
|
logger.debug("Connecting to Neuphonic")
|
||||||
|
|
||||||
tts_config = {
|
tts_config = {
|
||||||
"lang_code": self._settings.lang_code,
|
"lang_code": self._settings.language,
|
||||||
"speed": self._settings.speed,
|
"speed": self._settings.speed,
|
||||||
"encoding": self._settings.encoding,
|
"encoding": self._settings.encoding,
|
||||||
"sampling_rate": self._settings.sampling_rate,
|
"sampling_rate": self._settings.sampling_rate,
|
||||||
@@ -298,7 +296,7 @@ class NeuphonicTTSService(InterruptibleTTSService):
|
|||||||
if value is not None:
|
if value is not None:
|
||||||
query_params.append(f"{key}={value}")
|
query_params.append(f"{key}={value}")
|
||||||
|
|
||||||
url = f"{self._url}/speak/{self._settings.lang_code}"
|
url = f"{self._url}/speak/{self._settings.language}"
|
||||||
if query_params:
|
if query_params:
|
||||||
url += f"?{'&'.join(query_params)}"
|
url += f"?{'&'.join(query_params)}"
|
||||||
|
|
||||||
@@ -407,6 +405,8 @@ class NeuphonicHttpTTSService(TTSService):
|
|||||||
HTTP-based communication over WebSocket connections.
|
HTTP-based communication over WebSocket connections.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
_settings: NeuphonicTTSSettings
|
||||||
|
|
||||||
class InputParams(BaseModel):
|
class InputParams(BaseModel):
|
||||||
"""Input parameters for Neuphonic HTTP TTS configuration.
|
"""Input parameters for Neuphonic HTTP TTS configuration.
|
||||||
|
|
||||||
@@ -449,10 +449,13 @@ class NeuphonicHttpTTSService(TTSService):
|
|||||||
self._api_key = api_key
|
self._api_key = api_key
|
||||||
self._session = aiohttp_session
|
self._session = aiohttp_session
|
||||||
self._base_url = url.rstrip("/")
|
self._base_url = url.rstrip("/")
|
||||||
self._lang_code = self.language_to_service_language(params.language) or "en"
|
self._settings = NeuphonicTTSSettings(
|
||||||
self._speed = params.speed
|
voice=voice_id,
|
||||||
self._encoding = encoding
|
language=self.language_to_service_language(params.language) or "en",
|
||||||
self._voice_id = voice_id
|
speed=params.speed,
|
||||||
|
encoding=encoding,
|
||||||
|
sampling_rate=sample_rate,
|
||||||
|
)
|
||||||
|
|
||||||
def can_generate_metrics(self) -> bool:
|
def can_generate_metrics(self) -> bool:
|
||||||
"""Check if this service can generate processing metrics.
|
"""Check if this service can generate processing metrics.
|
||||||
@@ -536,7 +539,7 @@ class NeuphonicHttpTTSService(TTSService):
|
|||||||
"""
|
"""
|
||||||
logger.debug(f"Generating TTS: [{text}]")
|
logger.debug(f"Generating TTS: [{text}]")
|
||||||
|
|
||||||
url = f"{self._base_url}/sse/speak/{self._lang_code}"
|
url = f"{self._base_url}/sse/speak/{self._settings.language}"
|
||||||
|
|
||||||
headers = {
|
headers = {
|
||||||
"X-API-KEY": self._api_key,
|
"X-API-KEY": self._api_key,
|
||||||
@@ -545,14 +548,14 @@ class NeuphonicHttpTTSService(TTSService):
|
|||||||
|
|
||||||
payload = {
|
payload = {
|
||||||
"text": text,
|
"text": text,
|
||||||
"lang_code": self._lang_code,
|
"lang_code": self._settings.language,
|
||||||
"encoding": self._encoding,
|
"encoding": self._settings.encoding,
|
||||||
"sampling_rate": self.sample_rate,
|
"sampling_rate": self.sample_rate,
|
||||||
"speed": self._speed,
|
"speed": self._settings.speed,
|
||||||
}
|
}
|
||||||
|
|
||||||
if self._voice_id:
|
if self._settings.voice:
|
||||||
payload["voice_id"] = self._voice_id
|
payload["voice_id"] = self._settings.voice
|
||||||
|
|
||||||
try:
|
try:
|
||||||
await self.start_ttfb_metrics()
|
await self.start_ttfb_metrics()
|
||||||
|
|||||||
@@ -12,7 +12,8 @@ gRPC API for high-quality speech synthesis.
|
|||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
import os
|
import os
|
||||||
from typing import AsyncGenerator, AsyncIterator, Generator, Mapping, Optional
|
from dataclasses import dataclass, field
|
||||||
|
from typing import Any, AsyncGenerator, AsyncIterator, Generator, Mapping, Optional
|
||||||
|
|
||||||
from pipecat.utils.tracing.service_decorators import traced_tts
|
from pipecat.utils.tracing.service_decorators import traced_tts
|
||||||
|
|
||||||
@@ -30,6 +31,7 @@ from pipecat.frames.frames import (
|
|||||||
TTSStartedFrame,
|
TTSStartedFrame,
|
||||||
TTSStoppedFrame,
|
TTSStoppedFrame,
|
||||||
)
|
)
|
||||||
|
from pipecat.services.settings import NOT_GIVEN, TTSSettings, _NotGiven
|
||||||
from pipecat.services.tts_service import TTSService
|
from pipecat.services.tts_service import TTSService
|
||||||
from pipecat.transcriptions.language import Language
|
from pipecat.transcriptions.language import Language
|
||||||
|
|
||||||
@@ -42,6 +44,17 @@ except ModuleNotFoundError as e:
|
|||||||
raise Exception(f"Missing module: {e}")
|
raise Exception(f"Missing module: {e}")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class NvidiaTTSSettings(TTSSettings):
|
||||||
|
"""Settings for NVIDIA Riva TTS service.
|
||||||
|
|
||||||
|
Parameters:
|
||||||
|
quality: Audio quality setting (0-100).
|
||||||
|
"""
|
||||||
|
|
||||||
|
quality: int | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
||||||
|
|
||||||
|
|
||||||
class NvidiaTTSService(TTSService):
|
class NvidiaTTSService(TTSService):
|
||||||
"""NVIDIA Riva text-to-speech service.
|
"""NVIDIA Riva text-to-speech service.
|
||||||
|
|
||||||
@@ -50,6 +63,8 @@ class NvidiaTTSService(TTSService):
|
|||||||
configurable quality settings.
|
configurable quality settings.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
_settings: NvidiaTTSSettings
|
||||||
|
|
||||||
class InputParams(BaseModel):
|
class InputParams(BaseModel):
|
||||||
"""Input parameters for Riva TTS configuration.
|
"""Input parameters for Riva TTS configuration.
|
||||||
|
|
||||||
@@ -94,13 +109,14 @@ class NvidiaTTSService(TTSService):
|
|||||||
|
|
||||||
self._server = server
|
self._server = server
|
||||||
self._api_key = api_key
|
self._api_key = api_key
|
||||||
self._voice_id = voice_id
|
|
||||||
self._language_code = params.language
|
|
||||||
self._quality = params.quality
|
|
||||||
self._function_id = model_function_map.get("function_id")
|
self._function_id = model_function_map.get("function_id")
|
||||||
self._use_ssl = use_ssl
|
self._use_ssl = use_ssl
|
||||||
|
self._settings = NvidiaTTSSettings(
|
||||||
|
voice=voice_id,
|
||||||
|
language=params.language,
|
||||||
|
quality=params.quality,
|
||||||
|
)
|
||||||
self.set_model_name(model_function_map.get("model_name"))
|
self.set_model_name(model_function_map.get("model_name"))
|
||||||
self._voice_id = voice_id
|
|
||||||
|
|
||||||
self._service = None
|
self._service = None
|
||||||
self._config = None
|
self._config = None
|
||||||
@@ -133,6 +149,18 @@ class NvidiaTTSService(TTSService):
|
|||||||
stacklevel=2,
|
stacklevel=2,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
async def _update_settings(self, update: NvidiaTTSSettings) -> dict[str, Any]:
|
||||||
|
"""Apply a settings update.
|
||||||
|
|
||||||
|
Settings are stored but not applied to the active connection.
|
||||||
|
"""
|
||||||
|
changed = await super()._update_settings(update)
|
||||||
|
if not changed:
|
||||||
|
return changed
|
||||||
|
# TODO: reconnect gRPC client to apply changed settings.
|
||||||
|
self._warn_unhandled_updated_settings(changed)
|
||||||
|
return changed
|
||||||
|
|
||||||
def _initialize_client(self):
|
def _initialize_client(self):
|
||||||
if self._service is not None:
|
if self._service is not None:
|
||||||
return
|
return
|
||||||
@@ -181,11 +209,11 @@ class NvidiaTTSService(TTSService):
|
|||||||
def read_audio_responses() -> Generator[rtts.SynthesizeSpeechResponse, None, None]:
|
def read_audio_responses() -> Generator[rtts.SynthesizeSpeechResponse, None, None]:
|
||||||
responses = self._service.synthesize_online(
|
responses = self._service.synthesize_online(
|
||||||
text,
|
text,
|
||||||
self._voice_id,
|
self._settings.voice,
|
||||||
self._language_code,
|
self._settings.language,
|
||||||
sample_rate_hz=self.sample_rate,
|
sample_rate_hz=self.sample_rate,
|
||||||
zero_shot_audio_prompt_file=None,
|
zero_shot_audio_prompt_file=None,
|
||||||
zero_shot_quality=self._quality,
|
zero_shot_quality=self._settings.quality,
|
||||||
custom_dictionary={},
|
custom_dictionary={},
|
||||||
)
|
)
|
||||||
return responses
|
return responses
|
||||||
|
|||||||
@@ -7,8 +7,9 @@
|
|||||||
"""Piper TTS service implementation."""
|
"""Piper TTS service implementation."""
|
||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import AsyncGenerator, AsyncIterator, Optional
|
from typing import Any, AsyncGenerator, AsyncIterator, Optional
|
||||||
|
|
||||||
import aiohttp
|
import aiohttp
|
||||||
from loguru import logger
|
from loguru import logger
|
||||||
@@ -19,6 +20,7 @@ from pipecat.frames.frames import (
|
|||||||
TTSStartedFrame,
|
TTSStartedFrame,
|
||||||
TTSStoppedFrame,
|
TTSStoppedFrame,
|
||||||
)
|
)
|
||||||
|
from pipecat.services.settings import TTSSettings
|
||||||
from pipecat.services.tts_service import TTSService
|
from pipecat.services.tts_service import TTSService
|
||||||
from pipecat.utils.tracing.service_decorators import traced_tts
|
from pipecat.utils.tracing.service_decorators import traced_tts
|
||||||
|
|
||||||
@@ -31,6 +33,13 @@ except ModuleNotFoundError as e:
|
|||||||
raise Exception(f"Missing module: {e}")
|
raise Exception(f"Missing module: {e}")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class PiperTTSSettings(TTSSettings):
|
||||||
|
"""Settings for Piper TTS service."""
|
||||||
|
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
class PiperTTSService(TTSService):
|
class PiperTTSService(TTSService):
|
||||||
"""Piper TTS service implementation.
|
"""Piper TTS service implementation.
|
||||||
|
|
||||||
@@ -39,6 +48,8 @@ class PiperTTSService(TTSService):
|
|||||||
match the configured sample rate.
|
match the configured sample rate.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
_settings: PiperTTSSettings
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
@@ -60,7 +71,7 @@ class PiperTTSService(TTSService):
|
|||||||
"""
|
"""
|
||||||
super().__init__(**kwargs)
|
super().__init__(**kwargs)
|
||||||
|
|
||||||
self._voice_id = voice_id
|
self._settings = PiperTTSSettings(voice=voice_id)
|
||||||
|
|
||||||
download_dir = download_dir or Path.cwd()
|
download_dir = download_dir or Path.cwd()
|
||||||
|
|
||||||
@@ -85,6 +96,18 @@ class PiperTTSService(TTSService):
|
|||||||
"""
|
"""
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
async def _update_settings(self, update: PiperTTSSettings) -> dict[str, Any]:
|
||||||
|
"""Apply a settings update.
|
||||||
|
|
||||||
|
Settings are stored but not applied to the active connection.
|
||||||
|
"""
|
||||||
|
changed = await super()._update_settings(update)
|
||||||
|
if not changed:
|
||||||
|
return changed
|
||||||
|
# TODO: voice changes would require re-downloading and loading the model.
|
||||||
|
self._warn_unhandled_updated_settings(changed)
|
||||||
|
return changed
|
||||||
|
|
||||||
@traced_tts
|
@traced_tts
|
||||||
async def run_tts(self, text: str, context_id: str) -> AsyncGenerator[Frame, None]:
|
async def run_tts(self, text: str, context_id: str) -> AsyncGenerator[Frame, None]:
|
||||||
"""Generate speech from text using Piper.
|
"""Generate speech from text using Piper.
|
||||||
@@ -143,6 +166,13 @@ class PiperTTSService(TTSService):
|
|||||||
# $ uv pip install "piper-tts[http]"
|
# $ uv pip install "piper-tts[http]"
|
||||||
# $ uv run python -m piper.http_server -m en_US-ryan-high
|
# $ uv run python -m piper.http_server -m en_US-ryan-high
|
||||||
#
|
#
|
||||||
|
@dataclass
|
||||||
|
class PiperHttpTTSSettings(TTSSettings):
|
||||||
|
"""Settings for Piper HTTP TTS service."""
|
||||||
|
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
class PiperHttpTTSService(TTSService):
|
class PiperHttpTTSService(TTSService):
|
||||||
"""Piper HTTP TTS service implementation.
|
"""Piper HTTP TTS service implementation.
|
||||||
|
|
||||||
@@ -151,6 +181,8 @@ class PiperHttpTTSService(TTSService):
|
|||||||
rates and automatic WAV header removal.
|
rates and automatic WAV header removal.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
_settings: PiperHttpTTSSettings
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
@@ -175,7 +207,7 @@ class PiperHttpTTSService(TTSService):
|
|||||||
|
|
||||||
self._base_url = base_url
|
self._base_url = base_url
|
||||||
self._session = aiohttp_session
|
self._session = aiohttp_session
|
||||||
self._model_id = voice_id
|
self._settings = PiperHttpTTSSettings(voice=voice_id)
|
||||||
|
|
||||||
def can_generate_metrics(self) -> bool:
|
def can_generate_metrics(self) -> bool:
|
||||||
"""Check if this service can generate processing metrics.
|
"""Check if this service can generate processing metrics.
|
||||||
@@ -205,7 +237,7 @@ class PiperHttpTTSService(TTSService):
|
|||||||
|
|
||||||
data = {
|
data = {
|
||||||
"text": text,
|
"text": text,
|
||||||
"voice": self._model_id,
|
"voice": self._settings.voice,
|
||||||
}
|
}
|
||||||
|
|
||||||
async with self._session.post(self._base_url, json=data, headers=headers) as response:
|
async with self._session.post(self._base_url, json=data, headers=headers) as response:
|
||||||
|
|||||||
@@ -7,7 +7,8 @@
|
|||||||
"""Speechmatics TTS service integration."""
|
"""Speechmatics TTS service integration."""
|
||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
from typing import AsyncGenerator, Optional
|
from dataclasses import dataclass, field
|
||||||
|
from typing import Any, AsyncGenerator, Optional
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
|
|
||||||
import aiohttp
|
import aiohttp
|
||||||
@@ -21,6 +22,7 @@ from pipecat.frames.frames import (
|
|||||||
TTSStartedFrame,
|
TTSStartedFrame,
|
||||||
TTSStoppedFrame,
|
TTSStoppedFrame,
|
||||||
)
|
)
|
||||||
|
from pipecat.services.settings import NOT_GIVEN, TTSSettings, _NotGiven
|
||||||
from pipecat.services.tts_service import TTSService
|
from pipecat.services.tts_service import TTSService
|
||||||
from pipecat.utils.network import exponential_backoff_time
|
from pipecat.utils.network import exponential_backoff_time
|
||||||
from pipecat.utils.tracing.service_decorators import traced_tts
|
from pipecat.utils.tracing.service_decorators import traced_tts
|
||||||
@@ -35,6 +37,17 @@ except ModuleNotFoundError as e:
|
|||||||
raise Exception(f"Missing module: {e}")
|
raise Exception(f"Missing module: {e}")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class SpeechmaticsTTSSettings(TTSSettings):
|
||||||
|
"""Settings for Speechmatics TTS service.
|
||||||
|
|
||||||
|
Parameters:
|
||||||
|
max_retries: Maximum number of retries for HTTP requests.
|
||||||
|
"""
|
||||||
|
|
||||||
|
max_retries: int | _NotGiven = field(default_factory=lambda: NOT_GIVEN)
|
||||||
|
|
||||||
|
|
||||||
class SpeechmaticsTTSService(TTSService):
|
class SpeechmaticsTTSService(TTSService):
|
||||||
"""Speechmatics TTS service implementation.
|
"""Speechmatics TTS service implementation.
|
||||||
|
|
||||||
@@ -42,6 +55,8 @@ class SpeechmaticsTTSService(TTSService):
|
|||||||
It converts text to speech and returns raw PCM audio data for real-time playback.
|
It converts text to speech and returns raw PCM audio data for real-time playback.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
_settings: SpeechmaticsTTSSettings
|
||||||
|
|
||||||
SPEECHMATICS_SAMPLE_RATE = 16000
|
SPEECHMATICS_SAMPLE_RATE = 16000
|
||||||
|
|
||||||
class InputParams(BaseModel):
|
class InputParams(BaseModel):
|
||||||
@@ -91,11 +106,11 @@ class SpeechmaticsTTSService(TTSService):
|
|||||||
if not self._api_key:
|
if not self._api_key:
|
||||||
raise ValueError("Missing Speechmatics API key")
|
raise ValueError("Missing Speechmatics API key")
|
||||||
|
|
||||||
# Default parameters
|
params = params or SpeechmaticsTTSService.InputParams()
|
||||||
self._params = params or SpeechmaticsTTSService.InputParams()
|
self._settings = SpeechmaticsTTSSettings(
|
||||||
|
voice=voice_id,
|
||||||
# Set voice from constructor parameter
|
max_retries=params.max_retries,
|
||||||
self._voice_id = voice_id
|
)
|
||||||
|
|
||||||
def can_generate_metrics(self) -> bool:
|
def can_generate_metrics(self) -> bool:
|
||||||
"""Check if this service can generate processing metrics.
|
"""Check if this service can generate processing metrics.
|
||||||
@@ -131,7 +146,7 @@ class SpeechmaticsTTSService(TTSService):
|
|||||||
}
|
}
|
||||||
|
|
||||||
# Complete HTTP URL
|
# Complete HTTP URL
|
||||||
url = _get_endpoint_url(self._base_url, self._voice_id, self.sample_rate)
|
url = _get_endpoint_url(self._base_url, self._settings.voice, self.sample_rate)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
# Start TTS TTFB metrics
|
# Start TTS TTFB metrics
|
||||||
@@ -159,7 +174,7 @@ class SpeechmaticsTTSService(TTSService):
|
|||||||
attempt += 1
|
attempt += 1
|
||||||
|
|
||||||
# Check if we've exceeded the maximum number of attempts
|
# Check if we've exceeded the maximum number of attempts
|
||||||
if attempt >= self._params.max_retries:
|
if attempt >= self._settings.max_retries:
|
||||||
raise ValueError()
|
raise ValueError()
|
||||||
|
|
||||||
# Report error frame
|
# Report error frame
|
||||||
|
|||||||
Reference in New Issue
Block a user