Merge pull request #1500 from pipecat-ai/aleix/base-output-transport-optimize-bot-speaking
BaseOutputTransport: optimize BotSpeakingFrames
This commit is contained in:
27
CHANGELOG.md
27
CHANGELOG.md
@@ -76,17 +76,6 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
- `GladiaSTTService` now uses the `solaria-1` model by default. Other params
|
- `GladiaSTTService` now uses the `solaria-1` model by default. Other params
|
||||||
use Gladia's default values. Added support for more language codes.
|
use Gladia's default values. Added support for more language codes.
|
||||||
|
|
||||||
### Fixed
|
|
||||||
|
|
||||||
- Fixed an issue that could cause the `TranscriptionUpdateFrame` being pushed
|
|
||||||
because of an interruption to be discarded.
|
|
||||||
|
|
||||||
- Fixed an issue that would cause `SegmentedSTTService` based services
|
|
||||||
(e.g. `OpenAISTTService`) to try to transcribe non-spoken audio, causing
|
|
||||||
invalid transcriptions.
|
|
||||||
|
|
||||||
- Fixed an issue where `GoogleTTSService` was emitting two `TTSStoppedFrames`.
|
|
||||||
|
|
||||||
### Deprecated
|
### Deprecated
|
||||||
|
|
||||||
- All Pipecat services imports have been deprecated and a warning will be shown
|
- All Pipecat services imports have been deprecated and a warning will be shown
|
||||||
@@ -104,6 +93,22 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
- Deprecated using `GladiaSTTService.InputParams` directly. Use the new
|
- Deprecated using `GladiaSTTService.InputParams` directly. Use the new
|
||||||
`GladiaInputParams` class instead.
|
`GladiaInputParams` class instead.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- Fixed an issue that could cause the `TranscriptionUpdateFrame` being pushed
|
||||||
|
because of an interruption to be discarded.
|
||||||
|
|
||||||
|
- Fixed an issue that would cause `SegmentedSTTService` based services
|
||||||
|
(e.g. `OpenAISTTService`) to try to transcribe non-spoken audio, causing
|
||||||
|
invalid transcriptions.
|
||||||
|
|
||||||
|
- Fixed an issue where `GoogleTTSService` was emitting two `TTSStoppedFrames`.
|
||||||
|
|
||||||
|
### Performance
|
||||||
|
|
||||||
|
- `BotSpeakingFrame`s are now sent every 200ms. If the output transport audio chunks
|
||||||
|
are higher than 200ms then they will be sent at every audio chunk.
|
||||||
|
|
||||||
### Other
|
### Other
|
||||||
|
|
||||||
- Added foundational example `37-mem0.py` demonstrating how to use the
|
- Added foundational example `37-mem0.py` demonstrating how to use the
|
||||||
|
|||||||
@@ -336,13 +336,22 @@ class BaseOutputTransport(FrameProcessor):
|
|||||||
return without_mixer(BOT_VAD_STOP_SECS)
|
return without_mixer(BOT_VAD_STOP_SECS)
|
||||||
|
|
||||||
async def _sink_task_handler(self):
|
async def _sink_task_handler(self):
|
||||||
|
# Push a BotSpeakingFrame every 200ms, we don't really need to push it
|
||||||
|
# at every audio chunk. If the audio chunk is bigger than 200ms, push at
|
||||||
|
# every audio chunk.
|
||||||
|
TOTAL_CHUNK_MS = self._params.audio_out_10ms_chunks * 10
|
||||||
|
BOT_SPEAKING_CHUNK_PERIOD = max(int(200 / TOTAL_CHUNK_MS), 1)
|
||||||
|
bot_speaking_counter = 0
|
||||||
async for frame in self._next_frame():
|
async for frame in self._next_frame():
|
||||||
# Notify the bot started speaking upstream if necessary and that
|
# Notify the bot started speaking upstream if necessary and that
|
||||||
# it's actually speaking.
|
# it's actually speaking.
|
||||||
if isinstance(frame, TTSAudioRawFrame):
|
if isinstance(frame, TTSAudioRawFrame):
|
||||||
await self._bot_started_speaking()
|
await self._bot_started_speaking()
|
||||||
await self.push_frame(BotSpeakingFrame())
|
if bot_speaking_counter % BOT_SPEAKING_CHUNK_PERIOD == 0:
|
||||||
await self.push_frame(BotSpeakingFrame(), FrameDirection.UPSTREAM)
|
await self.push_frame(BotSpeakingFrame())
|
||||||
|
await self.push_frame(BotSpeakingFrame(), FrameDirection.UPSTREAM)
|
||||||
|
bot_speaking_counter = 0
|
||||||
|
bot_speaking_counter += 1
|
||||||
|
|
||||||
# No need to push EndFrame, it's pushed from process_frame().
|
# No need to push EndFrame, it's pushed from process_frame().
|
||||||
if isinstance(frame, EndFrame):
|
if isinstance(frame, EndFrame):
|
||||||
|
|||||||
Reference in New Issue
Block a user