docs: hold music demo
This commit is contained in:
@@ -8,6 +8,7 @@ import argparse
|
|||||||
import asyncio
|
import asyncio
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
|
||||||
import aiohttp
|
import aiohttp
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
@@ -16,11 +17,26 @@ from runner import configure_with_args
|
|||||||
|
|
||||||
from pipecat.audio.mixers.soundfile_mixer import SoundfileMixer
|
from pipecat.audio.mixers.soundfile_mixer import SoundfileMixer
|
||||||
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
||||||
from pipecat.frames.frames import MixerEnableFrame, MixerUpdateSettingsFrame
|
from pipecat.frames.frames import (
|
||||||
|
BotInterruptionFrame,
|
||||||
|
BotSpeakingFrame,
|
||||||
|
BotStartedSpeakingFrame,
|
||||||
|
BotStoppedSpeakingFrame,
|
||||||
|
ControlFrame,
|
||||||
|
Frame,
|
||||||
|
InputAudioRawFrame,
|
||||||
|
LLMTextFrame,
|
||||||
|
MetricsFrame,
|
||||||
|
MixerEnableFrame,
|
||||||
|
MixerUpdateSettingsFrame,
|
||||||
|
TextFrame,
|
||||||
|
TTSAudioRawFrame,
|
||||||
|
)
|
||||||
from pipecat.pipeline.pipeline import Pipeline
|
from pipecat.pipeline.pipeline import Pipeline
|
||||||
from pipecat.pipeline.runner import PipelineRunner
|
from pipecat.pipeline.runner import PipelineRunner
|
||||||
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
||||||
from pipecat.processors.aggregators.openai_llm_context import OpenAILLMContext
|
from pipecat.processors.aggregators.openai_llm_context import OpenAILLMContext
|
||||||
|
from pipecat.processors.frame_processor import FrameDirection, FrameProcessor
|
||||||
from pipecat.services.cartesia import CartesiaTTSService
|
from pipecat.services.cartesia import CartesiaTTSService
|
||||||
from pipecat.services.openai import OpenAILLMService
|
from pipecat.services.openai import OpenAILLMService
|
||||||
from pipecat.transports.services.daily import DailyParams, DailyTransport
|
from pipecat.transports.services.daily import DailyParams, DailyTransport
|
||||||
@@ -28,10 +44,63 @@ from pipecat.transports.services.daily import DailyParams, DailyTransport
|
|||||||
load_dotenv(override=True)
|
load_dotenv(override=True)
|
||||||
|
|
||||||
logger.remove(0)
|
logger.remove(0)
|
||||||
logger.add(sys.stderr, level="DEBUG")
|
logger.add(sys.stderr, level="INFO")
|
||||||
|
|
||||||
|
|
||||||
|
class DebugProcessor(FrameProcessor):
|
||||||
|
"""A processor for debugging frames in the pipeline."""
|
||||||
|
|
||||||
|
def __init__(self, name, **kwargs): # noqa: D107
|
||||||
|
self._name = name
|
||||||
|
super().__init__(**kwargs)
|
||||||
|
|
||||||
|
async def process_frame(self, frame: Frame, direction: FrameDirection): # noqa: D102
|
||||||
|
await super().process_frame(frame, direction)
|
||||||
|
if not (
|
||||||
|
isinstance(frame, InputAudioRawFrame)
|
||||||
|
or isinstance(frame, TTSAudioRawFrame)
|
||||||
|
or isinstance(frame, BotSpeakingFrame)
|
||||||
|
or isinstance(frame, BotStartedSpeakingFrame)
|
||||||
|
or isinstance(frame, MetricsFrame)
|
||||||
|
or isinstance(frame, LLMTextFrame)
|
||||||
|
):
|
||||||
|
logger.info(f"{self._name}: {frame} {direction}")
|
||||||
|
await self.push_frame(frame, direction)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class StartHoldMusicFrame(ControlFrame):
|
||||||
|
"""Starts hold music."""
|
||||||
|
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
class HoldMusicProcessor(FrameProcessor):
|
||||||
|
"""A processor to play hold music."""
|
||||||
|
|
||||||
|
def __init__(self, **kwargs): # noqa: D107
|
||||||
|
super().__init__(**kwargs)
|
||||||
|
self._play_hold_music = False
|
||||||
|
|
||||||
|
async def process_frame(self, frame: Frame, direction: FrameDirection): # noqa: D102
|
||||||
|
await super().process_frame(frame, direction)
|
||||||
|
|
||||||
|
if isinstance(frame, StartHoldMusicFrame):
|
||||||
|
self._play_hold_music = True
|
||||||
|
|
||||||
|
if isinstance(frame, BotStoppedSpeakingFrame) and self._play_hold_music:
|
||||||
|
await self.push_frame(
|
||||||
|
MixerUpdateSettingsFrame({"volume": 1, "sound": "office", "loop": False})
|
||||||
|
)
|
||||||
|
await self.push_frame(MixerEnableFrame(True))
|
||||||
|
# await self.queue_frame(BotInterruptionFrame(), FrameDirection.UPSTREAM)
|
||||||
|
elif isinstance(frame, BotSpeakingFrame):
|
||||||
|
await self.push_frame(MixerEnableFrame(False))
|
||||||
|
await self.push_frame(frame, direction)
|
||||||
|
|
||||||
|
|
||||||
async def main():
|
async def main():
|
||||||
|
"""Main function to run the bot background sound."""
|
||||||
async with aiohttp.ClientSession() as session:
|
async with aiohttp.ClientSession() as session:
|
||||||
parser = argparse.ArgumentParser(description="Bot Background Sound")
|
parser = argparse.ArgumentParser(description="Bot Background Sound")
|
||||||
parser.add_argument("-i", "--input", type=str, required=True, help="Input audio file")
|
parser.add_argument("-i", "--input", type=str, required=True, help="Input audio file")
|
||||||
@@ -41,7 +110,7 @@ async def main():
|
|||||||
soundfile_mixer = SoundfileMixer(
|
soundfile_mixer = SoundfileMixer(
|
||||||
sound_files={"office": args.input},
|
sound_files={"office": args.input},
|
||||||
default_sound="office",
|
default_sound="office",
|
||||||
volume=2.0,
|
volume=0,
|
||||||
)
|
)
|
||||||
|
|
||||||
transport = DailyTransport(
|
transport = DailyTransport(
|
||||||
@@ -73,12 +142,16 @@ async def main():
|
|||||||
|
|
||||||
context = OpenAILLMContext(messages)
|
context = OpenAILLMContext(messages)
|
||||||
context_aggregator = llm.create_context_aggregator(context)
|
context_aggregator = llm.create_context_aggregator(context)
|
||||||
|
dp = DebugProcessor("post-llm")
|
||||||
|
hold_music_processor = HoldMusicProcessor()
|
||||||
|
|
||||||
pipeline = Pipeline(
|
pipeline = Pipeline(
|
||||||
[
|
[
|
||||||
transport.input(), # Transport user input
|
transport.input(), # Transport user input
|
||||||
context_aggregator.user(), # User responses
|
context_aggregator.user(), # User responses
|
||||||
llm, # LLM
|
llm, # LLM
|
||||||
|
dp, # Debug processor
|
||||||
|
hold_music_processor, # Hold music
|
||||||
tts, # TTS
|
tts, # TTS
|
||||||
transport.output(), # Transport bot output
|
transport.output(), # Transport bot output
|
||||||
context_aggregator.assistant(), # Assistant spoken responses
|
context_aggregator.assistant(), # Assistant spoken responses
|
||||||
@@ -98,17 +171,12 @@ async def main():
|
|||||||
@transport.event_handler("on_first_participant_joined")
|
@transport.event_handler("on_first_participant_joined")
|
||||||
async def on_first_participant_joined(transport, participant):
|
async def on_first_participant_joined(transport, participant):
|
||||||
await transport.capture_participant_transcription(participant["id"])
|
await transport.capture_participant_transcription(participant["id"])
|
||||||
# Show how to use mixer control frames.
|
|
||||||
await asyncio.sleep(10.0)
|
|
||||||
await task.queue_frame(MixerUpdateSettingsFrame({"volume": 0.5}))
|
|
||||||
await asyncio.sleep(5.0)
|
|
||||||
await task.queue_frame(MixerEnableFrame(False))
|
|
||||||
await asyncio.sleep(5.0)
|
|
||||||
await task.queue_frame(MixerEnableFrame(True))
|
|
||||||
await asyncio.sleep(5.0)
|
|
||||||
# Kick off the conversation.
|
# Kick off the conversation.
|
||||||
messages.append({"role": "system", "content": "Please introduce yourself to the user."})
|
messages.append({"role": "system", "content": "Please introduce yourself to the user."})
|
||||||
await task.queue_frames([context_aggregator.user().get_context_frame()])
|
await task.queue_frames([context_aggregator.user().get_context_frame()])
|
||||||
|
await task.queue_frame(TextFrame("I'm going to play some hold music."))
|
||||||
|
await task.queue_frame(StartHoldMusicFrame())
|
||||||
|
|
||||||
runner = PipelineRunner()
|
runner = PipelineRunner()
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user