audio: rename BotBackgroundSound file and allow loading multiple files
This commit is contained in:
@@ -16,7 +16,7 @@ from pipecat.pipeline.pipeline import Pipeline
|
|||||||
from pipecat.pipeline.runner import PipelineRunner
|
from pipecat.pipeline.runner import PipelineRunner
|
||||||
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
||||||
from pipecat.processors.aggregators.openai_llm_context import OpenAILLMContext
|
from pipecat.processors.aggregators.openai_llm_context import OpenAILLMContext
|
||||||
from pipecat.processors.audio.background_sound import BotBackgroundSound
|
from pipecat.processors.audio.bot_background_sound import BotBackgroundSound
|
||||||
from pipecat.services.cartesia import CartesiaTTSService
|
from pipecat.services.cartesia import CartesiaTTSService
|
||||||
from pipecat.services.openai import OpenAILLMService
|
from pipecat.services.openai import OpenAILLMService
|
||||||
from pipecat.transports.services.daily import DailyParams, DailyTransport
|
from pipecat.transports.services.daily import DailyParams, DailyTransport
|
||||||
@@ -69,7 +69,9 @@ async def main():
|
|||||||
context = OpenAILLMContext(messages)
|
context = OpenAILLMContext(messages)
|
||||||
context_aggregator = llm.create_context_aggregator(context)
|
context_aggregator = llm.create_context_aggregator(context)
|
||||||
|
|
||||||
background_sound = BotBackgroundSound(file_name=args.input)
|
background_sound = BotBackgroundSound(
|
||||||
|
sound_files={"office": args.input}, default_sound="office"
|
||||||
|
)
|
||||||
|
|
||||||
pipeline = Pipeline(
|
pipeline = Pipeline(
|
||||||
[
|
[
|
||||||
|
|||||||
@@ -6,6 +6,8 @@
|
|||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
|
|
||||||
|
from typing import Any, Dict, Mapping
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
|
||||||
from pipecat.audio.utils import resample_audio
|
from pipecat.audio.utils import resample_audio
|
||||||
@@ -36,17 +38,19 @@ except ModuleNotFoundError as e:
|
|||||||
class BotBackgroundSound(FrameProcessor):
|
class BotBackgroundSound(FrameProcessor):
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
file_name: str,
|
sound_files: Mapping[str, str],
|
||||||
|
default_sound: str,
|
||||||
volume: float = 0.4,
|
volume: float = 0.4,
|
||||||
sample_rate: int = 24000,
|
sample_rate: int = 24000,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
super().__init__(**kwargs)
|
super().__init__(**kwargs)
|
||||||
self._file_name = file_name
|
self._sound_files = sound_files
|
||||||
self._volume = volume
|
self._volume = volume
|
||||||
self._sample_rate = sample_rate
|
self._sample_rate = sample_rate
|
||||||
|
|
||||||
self._sound = np.array([], dtype=np.int16)
|
self._sounds: Dict[str, Any] = {}
|
||||||
|
self._current_sound = default_sound
|
||||||
self._sound_pos = 0
|
self._sound_pos = 0
|
||||||
|
|
||||||
self._bot_speaking = False
|
self._bot_speaking = False
|
||||||
@@ -72,9 +76,20 @@ class BotBackgroundSound(FrameProcessor):
|
|||||||
await self.push_frame(frame, direction)
|
await self.push_frame(frame, direction)
|
||||||
|
|
||||||
async def _start(self):
|
async def _start(self):
|
||||||
|
for sound_name, file_name in self._sound_files.items():
|
||||||
|
await asyncio.to_thread(self._load_sound_file, sound_name, file_name)
|
||||||
|
|
||||||
|
self._audio_queue = asyncio.Queue()
|
||||||
|
self._audio_task = self.get_event_loop().create_task(self._audio_task_handler())
|
||||||
|
|
||||||
|
async def _stop(self):
|
||||||
|
self._audio_task.cancel()
|
||||||
|
await self._audio_task
|
||||||
|
|
||||||
|
def _load_sound_file(self, sound_name: str, file_name: str):
|
||||||
try:
|
try:
|
||||||
logger.debug(f"{self} loading background sound from {self._file_name}")
|
logger.debug(f"{self} loading background sound from {file_name}")
|
||||||
sound, sample_rate = sf.read(self._file_name, dtype="int16")
|
sound, sample_rate = sf.read(file_name, dtype="int16")
|
||||||
|
|
||||||
audio = sound.tobytes()
|
audio = sound.tobytes()
|
||||||
if sample_rate != self._sample_rate:
|
if sample_rate != self._sample_rate:
|
||||||
@@ -82,16 +97,9 @@ class BotBackgroundSound(FrameProcessor):
|
|||||||
audio = resample_audio(audio, sample_rate, self._sample_rate)
|
audio = resample_audio(audio, sample_rate, self._sample_rate)
|
||||||
|
|
||||||
# Convert from np to bytes again.
|
# Convert from np to bytes again.
|
||||||
self._sound = np.frombuffer(audio, dtype=np.int16)
|
self._sounds[sound_name] = np.frombuffer(audio, dtype=np.int16)
|
||||||
|
|
||||||
self._audio_queue = asyncio.Queue()
|
|
||||||
self._audio_task = self.get_event_loop().create_task(self._audio_task_handler())
|
|
||||||
except Exception as ex:
|
except Exception as ex:
|
||||||
logger.error(f"{self} unable to open file {self._file_name}")
|
logger.error(f"{self} unable to open file {file_name}")
|
||||||
|
|
||||||
async def _stop(self):
|
|
||||||
self._audio_task.cancel()
|
|
||||||
await self._audio_task
|
|
||||||
|
|
||||||
def _mix_with_sound(self, audio: bytes):
|
def _mix_with_sound(self, audio: bytes):
|
||||||
"""Mixes raw audio frames with chunks of the same length from the sound
|
"""Mixes raw audio frames with chunks of the same length from the sound
|
||||||
@@ -106,15 +114,18 @@ class BotBackgroundSound(FrameProcessor):
|
|||||||
|
|
||||||
chunk_size = len(audio_np)
|
chunk_size = len(audio_np)
|
||||||
|
|
||||||
|
# Sound currently playing.
|
||||||
|
sound = self._sounds[self._current_sound]
|
||||||
|
|
||||||
# Go back to the beginning if we don't have enough data.
|
# Go back to the beginning if we don't have enough data.
|
||||||
if self._sound_pos + chunk_size > len(self._sound):
|
if self._sound_pos + chunk_size > len(sound):
|
||||||
self._sound_pos = 0
|
self._sound_pos = 0
|
||||||
|
|
||||||
start_pos = self._sound_pos
|
start_pos = self._sound_pos
|
||||||
end_pos = self._sound_pos + chunk_size
|
end_pos = self._sound_pos + chunk_size
|
||||||
self._sound_pos = end_pos
|
self._sound_pos = end_pos
|
||||||
|
|
||||||
sound_np = self._sound[start_pos:end_pos]
|
sound_np = sound[start_pos:end_pos]
|
||||||
|
|
||||||
mixed_audio = np.clip(audio_np + sound_np * self._volume, -32768, 32767).astype(np.int16)
|
mixed_audio = np.clip(audio_np + sound_np * self._volume, -32768, 32767).astype(np.int16)
|
||||||
|
|
||||||
Reference in New Issue
Block a user