Update OpenAIRealtimeLLMService to work with LLMContext and LLMContextAggregatorPair (cont'd).
Avoid pushing `LLMTextFrame` when `OpenAIRealtimeLLMService` is configured to output audio. This avoids duplicate text in assistant messages in context. Conceptually, a speech-to-speech service encapsulates TTS behavior; in a "traditional" pipeline, `LLMTextFrames` are swallowed by the TTS service, so they should similarly not be pushed by a speech-to-speech service. Only. `TTSTextFrame`s should be pushed.
This commit is contained in:
@@ -687,12 +687,15 @@ class OpenAIRealtimeLLMService(LLMService):
|
|||||||
logger.debug(f"Handling standalone response: {evt.response.id}")
|
logger.debug(f"Handling standalone response: {evt.response.id}")
|
||||||
|
|
||||||
async def _handle_evt_text_delta(self, evt):
|
async def _handle_evt_text_delta(self, evt):
|
||||||
|
# We receive text deltas (as opposed to audio transcript deltas) when
|
||||||
|
# the output modality is "text"
|
||||||
if evt.delta:
|
if evt.delta:
|
||||||
await self.push_frame(LLMTextFrame(evt.delta))
|
await self.push_frame(LLMTextFrame(evt.delta))
|
||||||
|
|
||||||
async def _handle_evt_audio_transcript_delta(self, evt):
|
async def _handle_evt_audio_transcript_delta(self, evt):
|
||||||
|
# We receive audio transcript deltas (as opposed to text deltas) when
|
||||||
|
# the output modality is "audio" (the default)
|
||||||
if evt.delta:
|
if evt.delta:
|
||||||
await self.push_frame(LLMTextFrame(evt.delta))
|
|
||||||
await self.push_frame(TTSTextFrame(evt.delta))
|
await self.push_frame(TTSTextFrame(evt.delta))
|
||||||
|
|
||||||
async def _handle_evt_function_call_arguments_done(self, evt):
|
async def _handle_evt_function_call_arguments_done(self, evt):
|
||||||
|
|||||||
Reference in New Issue
Block a user