Merge pull request #931 from pipecat-ai/mb/fix-openai-realtime-
Fix truncation timing of OpenAIRealtimeBetaLLMService
This commit is contained in:
@@ -14,6 +14,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
- Added `29-livekit-audio-chat.py`, as a new foundational examples for
|
- Added `29-livekit-audio-chat.py`, as a new foundational examples for
|
||||||
`LiveKitTransportLayer`.
|
`LiveKitTransportLayer`.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- Fixed an issue where `OpenAIRealtimeBetaLLMService` audio chunks were hitting
|
||||||
|
an error when truncating audio content.
|
||||||
|
|
||||||
## [0.0.52] - 2024-12-24
|
## [0.0.52] - 2024-12-24
|
||||||
|
|
||||||
### Added
|
### Added
|
||||||
|
|||||||
@@ -152,17 +152,38 @@ class OpenAIRealtimeBetaLLMService(LLMService):
|
|||||||
async def _handle_bot_stopped_speaking(self):
|
async def _handle_bot_stopped_speaking(self):
|
||||||
self._current_audio_response = None
|
self._current_audio_response = None
|
||||||
|
|
||||||
|
def _calculate_audio_duration_ms(
|
||||||
|
self, total_bytes: int, sample_rate: int = 24000, bytes_per_sample: int = 2
|
||||||
|
) -> int:
|
||||||
|
"""Calculate audio duration in milliseconds based on PCM audio parameters."""
|
||||||
|
samples = total_bytes / bytes_per_sample
|
||||||
|
duration_seconds = samples / sample_rate
|
||||||
|
return int(duration_seconds * 1000)
|
||||||
|
|
||||||
async def _truncate_current_audio_response(self):
|
async def _truncate_current_audio_response(self):
|
||||||
|
"""Truncates the current audio response at the appropriate duration.
|
||||||
|
|
||||||
|
Calculates the actual duration of the audio content and truncates at the shorter of
|
||||||
|
either the wall clock time or the actual audio duration to prevent invalid truncation
|
||||||
|
requests.
|
||||||
|
"""
|
||||||
# if the bot is still speaking, truncate the last message
|
# if the bot is still speaking, truncate the last message
|
||||||
if self._current_audio_response:
|
if self._current_audio_response:
|
||||||
current = self._current_audio_response
|
current = self._current_audio_response
|
||||||
self._current_audio_response = None
|
self._current_audio_response = None
|
||||||
|
|
||||||
|
# Calculate actual audio duration instead of using wall clock time
|
||||||
|
audio_duration_ms = self._calculate_audio_duration_ms(current.total_size)
|
||||||
|
|
||||||
|
# Use the shorter of wall clock time or actual audio duration
|
||||||
elapsed_ms = int(time.time() * 1000 - current.start_time_ms)
|
elapsed_ms = int(time.time() * 1000 - current.start_time_ms)
|
||||||
|
truncate_ms = min(elapsed_ms, audio_duration_ms)
|
||||||
|
|
||||||
await self.send_client_event(
|
await self.send_client_event(
|
||||||
events.ConversationItemTruncateEvent(
|
events.ConversationItemTruncateEvent(
|
||||||
item_id=current.item_id,
|
item_id=current.item_id,
|
||||||
content_index=current.content_index,
|
content_index=current.content_index,
|
||||||
audio_end_ms=elapsed_ms,
|
audio_end_ms=truncate_ms,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user