Migrate realtime examples to RealtimeServiceModeConfig
Pass realtime_service_mode=RealtimeServiceModeConfig() through every realtime LLM service example (base, async-tool, video, text-output, persistent-context, update-settings, MCP) so context aggregation uses the new realtime-mode semantics instead of relying on local VAD as a workaround. Where examples previously wired SileroVADAnalyzer into LLMUserAggregatorParams to coax turn frames out of services that don't emit them server-side (AWS Nova Sonic, Ultravox, Gemini Live), the local VAD is now removed. realtime_service_mode keeps context writes correct without it, and the Phase 1.5 server-side InterruptionFrame fixes for Nova Sonic and Ultravox keep the bot from talking past the user when they barge in. Transcript-logging event handlers move from on_user_turn_stopped / on_assistant_turn_stopped to on_user_message_added / on_assistant_message_added, which carry the finalized text in realtime mode (the turn-stopped events fire before the message is finalized, so their `content` is None in that mode). For services that don't emit user-turn frames (Gemini Live, AWS Nova Sonic, Ultravox) the example now carries a Tier 1 comment block that spells out which downstream processors won't activate, how to add local VAD if needed, and the caveat that locally-generated turn boundaries are a heuristic that may diverge from server-side ground truth. Adds examples/realtime/realtime-openai-local-vad.py, a new variant of the OpenAI Realtime example that disables OpenAI's server-side turn detection and drives turn boundaries locally — useful when you want a turn analyzer like LocalSmartTurnV3 to decide when the user is done speaking. Server-emitted turn frames are still preferred when available. The Gemini Live local-VAD variant already existed; it's been updated in place rather than rewritten.
This commit is contained in:
@@ -12,7 +12,6 @@ from loguru import logger
|
||||
|
||||
from pipecat.adapters.schemas.function_schema import FunctionSchema
|
||||
from pipecat.adapters.schemas.tools_schema import ToolsSchema
|
||||
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
||||
from pipecat.pipeline.pipeline import Pipeline
|
||||
from pipecat.pipeline.runner import PipelineRunner
|
||||
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
||||
@@ -20,7 +19,7 @@ from pipecat.processors.aggregators.llm_context import LLMContext
|
||||
from pipecat.processors.aggregators.llm_response_universal import (
|
||||
AssistantTurnStoppedMessage,
|
||||
LLMContextAggregatorPair,
|
||||
LLMUserAggregatorParams,
|
||||
RealtimeServiceModeConfig,
|
||||
UserTurnStoppedMessage,
|
||||
)
|
||||
from pipecat.runner.types import RunnerArguments
|
||||
@@ -30,8 +29,6 @@ from pipecat.services.ultravox.llm import OneShotInputParams, UltravoxRealtimeLL
|
||||
from pipecat.transports.base_transport import BaseTransport, TransportParams
|
||||
from pipecat.transports.daily.transport import DailyParams
|
||||
from pipecat.transports.websocket.fastapi import FastAPIWebsocketParams
|
||||
from pipecat.turns.user_stop import SpeechTimeoutUserTurnStopStrategy
|
||||
from pipecat.turns.user_turn_strategies import UserTurnStrategies
|
||||
|
||||
# Load environment variables
|
||||
load_dotenv(override=True)
|
||||
@@ -178,18 +175,23 @@ There is also a secret menu that changes daily. If the user asks about it, use t
|
||||
|
||||
context = LLMContext([])
|
||||
|
||||
# Necessary to complete the function call lifecycle in Pipecat and
|
||||
# to produce user and assistant turn stopped events.
|
||||
# Ultravox drives the conversation server-side. It does NOT emit
|
||||
# UserStartedSpeakingFrame / UserStoppedSpeakingFrame, so pipeline
|
||||
# processors that depend on those frames — RTVI client speech events,
|
||||
# TurnTrackingObserver, AudioBufferProcessor turn recording,
|
||||
# UserIdleController, user mute strategies, voicemail detector — won't
|
||||
# activate with this default setup. Context aggregation still works
|
||||
# with realtime_service_mode.
|
||||
#
|
||||
# To produce these frames locally, wire a VAD analyzer (e.g.
|
||||
# SileroVADAnalyzer) into LLMUserAggregatorParams. Caveat: locally-
|
||||
# generated turn boundaries are a heuristic and may not match
|
||||
# Ultravox's server-side turn decisions, which is what drives the
|
||||
# conversation; the two can drift apart in subtle ways especially
|
||||
# around interruptions and overlapping speech.
|
||||
user_aggregator, assistant_aggregator = LLMContextAggregatorPair(
|
||||
context,
|
||||
user_params=LLMUserAggregatorParams(
|
||||
user_turn_strategies=UserTurnStrategies(
|
||||
stop=[SpeechTimeoutUserTurnStopStrategy()],
|
||||
),
|
||||
# Set the VAD analyzer to create reliable TTFB measurements and
|
||||
# user stop events.
|
||||
vad_analyzer=SileroVADAnalyzer(),
|
||||
),
|
||||
realtime_service_mode=RealtimeServiceModeConfig(),
|
||||
)
|
||||
|
||||
# Build the pipeline
|
||||
@@ -224,14 +226,18 @@ There is also a secret menu that changes daily. If the user asks about it, use t
|
||||
logger.info(f"Client disconnected")
|
||||
await task.cancel()
|
||||
|
||||
@user_aggregator.event_handler("on_user_turn_stopped")
|
||||
async def on_user_turn_stopped(aggregator, strategy, message: UserTurnStoppedMessage):
|
||||
# Ultravox doesn't emit user-turn frames so on_user_turn_stopped
|
||||
# would never fire. The *_message_added events fire when messages are
|
||||
# written to context and carry the finalized content; use those for
|
||||
# transcript logging.
|
||||
@user_aggregator.event_handler("on_user_message_added")
|
||||
async def on_user_message_added(aggregator, message: UserTurnStoppedMessage):
|
||||
timestamp = f"[{message.timestamp}] " if message.timestamp else ""
|
||||
line = f"{timestamp}user: {message.content}"
|
||||
logger.info(f"Transcript: {line}")
|
||||
|
||||
@assistant_aggregator.event_handler("on_assistant_turn_stopped")
|
||||
async def on_assistant_turn_stopped(aggregator, message: AssistantTurnStoppedMessage):
|
||||
@assistant_aggregator.event_handler("on_assistant_message_added")
|
||||
async def on_assistant_message_added(aggregator, message: AssistantTurnStoppedMessage):
|
||||
timestamp = f"[{message.timestamp}] " if message.timestamp else ""
|
||||
line = f"{timestamp}assistant: {message.content}"
|
||||
logger.info(f"Transcript: {line}")
|
||||
|
||||
Reference in New Issue
Block a user