Merge pull request #1151 from pipecat-ai/aleix/base-output-transport-resample
BaseOutputTransport: resample incoming audio if needed
This commit is contained in:
@@ -9,6 +9,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
|
- Fixed an issue in `BaseOutputTransport` where incoming audio would not be
|
||||||
|
resampled to the desired output sample rate.
|
||||||
|
|
||||||
- Fixed an issue with the `TwilioFrameSerializer` and `TelnyxFrameSerializer`
|
- Fixed an issue with the `TwilioFrameSerializer` and `TelnyxFrameSerializer`
|
||||||
where `frame.audio_out_sample_rate` was incorrectly used in place of
|
where `frame.audio_out_sample_rate` was incorrectly used in place of
|
||||||
`frame.audio_in_sample_rate`, which caused audio input detection to fail when
|
`frame.audio_in_sample_rate`, which caused audio input detection to fail when
|
||||||
|
|||||||
@@ -13,6 +13,7 @@ from typing import AsyncGenerator, List
|
|||||||
from loguru import logger
|
from loguru import logger
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
|
from pipecat.audio.utils import create_default_resampler
|
||||||
from pipecat.audio.vad.vad_analyzer import VAD_STOP_SECS
|
from pipecat.audio.vad.vad_analyzer import VAD_STOP_SECS
|
||||||
from pipecat.frames.frames import (
|
from pipecat.frames.frames import (
|
||||||
BotSpeakingFrame,
|
BotSpeakingFrame,
|
||||||
@@ -59,6 +60,7 @@ class BaseOutputTransport(FrameProcessor):
|
|||||||
|
|
||||||
# Output sample rate. It will be initialized on StartFrame.
|
# Output sample rate. It will be initialized on StartFrame.
|
||||||
self._sample_rate = 0
|
self._sample_rate = 0
|
||||||
|
self._resampler = create_default_resampler()
|
||||||
|
|
||||||
# Chunk size that will be written. It will be computed on StartFrame
|
# Chunk size that will be written. It will be computed on StartFrame
|
||||||
self._audio_chunk_size = 0
|
self._audio_chunk_size = 0
|
||||||
@@ -188,12 +190,18 @@ class BaseOutputTransport(FrameProcessor):
|
|||||||
if not self._params.audio_out_enabled:
|
if not self._params.audio_out_enabled:
|
||||||
return
|
return
|
||||||
|
|
||||||
|
# We might need to resample if incoming audio doesn't match the
|
||||||
|
# transport sample rate.
|
||||||
|
resampled = await self._resampler.resample(
|
||||||
|
frame.audio, frame.sample_rate, self._sample_rate
|
||||||
|
)
|
||||||
|
|
||||||
cls = type(frame)
|
cls = type(frame)
|
||||||
self._audio_buffer.extend(frame.audio)
|
self._audio_buffer.extend(resampled)
|
||||||
while len(self._audio_buffer) >= self._audio_chunk_size:
|
while len(self._audio_buffer) >= self._audio_chunk_size:
|
||||||
chunk = cls(
|
chunk = cls(
|
||||||
bytes(self._audio_buffer[: self._audio_chunk_size]),
|
bytes(self._audio_buffer[: self._audio_chunk_size]),
|
||||||
sample_rate=frame.sample_rate,
|
sample_rate=self._sample_rate,
|
||||||
num_channels=frame.num_channels,
|
num_channels=frame.num_channels,
|
||||||
)
|
)
|
||||||
await self._sink_queue.put(chunk)
|
await self._sink_queue.put(chunk)
|
||||||
|
|||||||
Reference in New Issue
Block a user