frames: deprecate emulated field in UserStartedSpeakingFrame/UserStoppedSpeakingFrame

This commit is contained in:
Aleix Conchillo Flaqué
2025-12-18 16:01:53 -08:00
parent a9cca0b934
commit 169fc0b568
2 changed files with 16 additions and 9 deletions

View File

@@ -1093,15 +1093,17 @@ class StartInterruptionFrame(InterruptionFrame):
@dataclass @dataclass
class UserStartedSpeakingFrame(SystemFrame): class UserStartedSpeakingFrame(SystemFrame):
"""Frame indicating user has started speaking. """Frame indicating that the user turn has started.
Emitted by VAD to indicate that a user has started speaking. This can be Emitted when the user turn starts, which usually means that some
used for interruptions or other times when detecting that someone is transcriptions are already available.
speaking is more important than knowing what they're saying (as you will
get with a TranscriptionFrame).
Parameters: Parameters:
emulated: Whether this event was emulated rather than detected by VAD. emulated: Whether this event was emulated rather than detected by VAD.
.. deprecated:: 0.0.99
This field is deprecated and will be removed in a future version.
""" """
emulated: bool = False emulated: bool = False
@@ -1109,12 +1111,17 @@ class UserStartedSpeakingFrame(SystemFrame):
@dataclass @dataclass
class UserStoppedSpeakingFrame(SystemFrame): class UserStoppedSpeakingFrame(SystemFrame):
"""Frame indicating user has stopped speaking. """Frame indicating that the user turn has ended.
Emitted by the VAD to indicate that a user stopped speaking. Emitted when the user turn ends. This usually coincides with the start of
the bot turn.
Parameters: Parameters:
emulated: Whether this event was emulated rather than detected by VAD. emulated: Whether this event was emulated rather than detected by VAD.
.. deprecated:: 0.0.99
This field is deprecated and will be removed in a future version.
""" """
emulated: bool = False emulated: bool = False

View File

@@ -407,7 +407,7 @@ class LLMUserAggregator(LLMContextAggregator):
if self._params.enable_user_speaking_frames: if self._params.enable_user_speaking_frames:
logger.debug(f"User started speaking (user turn start strategy: {strategy})") logger.debug(f"User started speaking (user turn start strategy: {strategy})")
# TODO(aleix): These frames should really come from the top of the pipeline. # TODO(aleix): These frames should really come from the top of the pipeline.
await self.broadcast_frame(UserStartedSpeakingFrame, emulated=strategy is None) await self.broadcast_frame(UserStartedSpeakingFrame)
await self.broadcast_frame(InterruptionFrame) await self.broadcast_frame(InterruptionFrame)
async def _trigger_bot_turn_start(self, strategy: BaseBotTurnStartStrategy): async def _trigger_bot_turn_start(self, strategy: BaseBotTurnStartStrategy):
@@ -419,7 +419,7 @@ class LLMUserAggregator(LLMContextAggregator):
if self._params.enable_user_speaking_frames: if self._params.enable_user_speaking_frames:
logger.debug(f"User stopped speaking (bot turn start strategy: {strategy})") logger.debug(f"User stopped speaking (bot turn start strategy: {strategy})")
# TODO(aleix): This frame should really come from the top of the pipeline. # TODO(aleix): This frame should really come from the top of the pipeline.
await self.broadcast_frame(UserStoppedSpeakingFrame, emulated=strategy is None) await self.broadcast_frame(UserStoppedSpeakingFrame)
# Always push context frame. # Always push context frame.
await self.push_aggregation() await self.push_aggregation()