Merge branch 'main' into piper-tts

# Conflicts:
#	CHANGELOG.md
This commit is contained in:
Filipi Fuchter
2025-03-27 08:01:38 -03:00
5 changed files with 33 additions and 10 deletions

View File

@@ -12,6 +12,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- Added support for a new TTS service, `PiperTTSService`. - Added support for a new TTS service, `PiperTTSService`.
(see https://github.com/rhasspy/piper/) (see https://github.com/rhasspy/piper/)
- It is now possible to tell whether `UserStartedSpeakingFrame` or
`UserStoppedSpeakingFrame` have been generated because of emulation frames.
### Fixed
- Fixed an issue that would cause `SegmentedSTTService` based services
(e.g. `OpenAISTTService`) to try to transcribe non-spoken audio, causing
invalid transcriptions.
## [0.0.61] - 2025-03-26 ## [0.0.61] - 2025-03-26
### Added ### Added
@@ -28,9 +37,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- ElevenLabs TTS services now support a sample rate of 8000. - ElevenLabs TTS services now support a sample rate of 8000.
- Added support for `instructions` in `OpenAITTSService` - Added support for `instructions` in `OpenAITTSService`.
- Added support for `base_url` in `OpenAIImageGenService` and `OpenAITTSService` - Added support for `base_url` in `OpenAIImageGenService` and
`OpenAITTSService`.
### Fixed ### Fixed

View File

@@ -562,14 +562,14 @@ class UserStartedSpeakingFrame(SystemFrame):
""" """
pass emulated: bool = False
@dataclass @dataclass
class UserStoppedSpeakingFrame(SystemFrame): class UserStoppedSpeakingFrame(SystemFrame):
"""Emitted by the VAD to indicate that a user stopped speaking.""" """Emitted by the VAD to indicate that a user stopped speaking."""
pass emulated: bool = False
@dataclass @dataclass

View File

@@ -249,7 +249,7 @@ class LLMUserContextAggregator(LLMContextResponseAggregator):
self._waiting_for_aggregation = False self._waiting_for_aggregation = False
async def handle_aggregation(self, aggregation: str): async def handle_aggregation(self, aggregation: str):
self._context.add_message({"role": self.role, "content": self._aggregation}) self._context.add_message({"role": self.role, "content": aggregation})
async def process_frame(self, frame: Frame, direction: FrameDirection): async def process_frame(self, frame: Frame, direction: FrameDirection):
await super().process_frame(frame, direction) await super().process_frame(frame, direction)
@@ -290,12 +290,14 @@ class LLMUserContextAggregator(LLMContextResponseAggregator):
async def push_aggregation(self): async def push_aggregation(self):
if len(self._aggregation) > 0: if len(self._aggregation) > 0:
await self.handle_aggregation(self._aggregation) aggregation = self._aggregation
# Reset the aggregation. Reset it before pushing it down, otherwise # Reset the aggregation. Reset it before pushing it down, otherwise
# if the tasks gets cancelled we won't be able to clear things up. # if the tasks gets cancelled we won't be able to clear things up.
self.reset() self.reset()
await self.handle_aggregation(aggregation)
frame = OpenAILLMContextFrame(self._context) frame = OpenAILLMContextFrame(self._context)
await self.push_frame(frame) await self.push_frame(frame)
@@ -308,10 +310,16 @@ class LLMUserContextAggregator(LLMContextResponseAggregator):
async def _cancel(self, frame: CancelFrame): async def _cancel(self, frame: CancelFrame):
await self._cancel_aggregation_task() await self._cancel_aggregation_task()
async def _handle_user_started_speaking(self, _: UserStartedSpeakingFrame): async def _handle_user_started_speaking(self, frame: UserStartedSpeakingFrame):
self._user_speaking = True self._user_speaking = True
self._waiting_for_aggregation = True self._waiting_for_aggregation = True
# If we get a non-emulated UserStartedSpeakingFrame but we are in the
# middle of emulating VAD, let's stop emulating VAD (i.e. don't send the
# EmulateUserStoppedSpeakingFrame).
if not frame.emulated and self._emulating_vad:
self._emulating_vad = False
async def _handle_user_stopped_speaking(self, _: UserStoppedSpeakingFrame): async def _handle_user_stopped_speaking(self, _: UserStoppedSpeakingFrame):
self._user_speaking = False self._user_speaking = False
# We just stopped speaking. Let's see if there's some aggregation to # We just stopped speaking. Let's see if there's some aggregation to

View File

@@ -1048,9 +1048,14 @@ class SegmentedSTTService(STTService):
await self._handle_user_stopped_speaking(frame) await self._handle_user_stopped_speaking(frame)
async def _handle_user_started_speaking(self, frame: UserStartedSpeakingFrame): async def _handle_user_started_speaking(self, frame: UserStartedSpeakingFrame):
if frame.emulated:
return
self._user_speaking = True self._user_speaking = True
async def _handle_user_stopped_speaking(self, frame: UserStoppedSpeakingFrame): async def _handle_user_stopped_speaking(self, frame: UserStoppedSpeakingFrame):
if frame.emulated:
return
self._user_speaking = False self._user_speaking = False
content = io.BytesIO() content = io.BytesIO()
@@ -1068,7 +1073,7 @@ class SegmentedSTTService(STTService):
self._audio_buffer.clear() self._audio_buffer.clear()
async def process_audio_frame(self, frame: AudioRawFrame, direction: FrameDirection): async def process_audio_frame(self, frame: AudioRawFrame, direction: FrameDirection):
# If the user is speaking the audio buffer will keep growin. # If the user is speaking the audio buffer will keep growing.
self._audio_buffer += frame.audio self._audio_buffer += frame.audio
# If the user is not speaking we keep just a little bit of audio. # If the user is not speaking we keep just a little bit of audio.

View File

@@ -117,10 +117,10 @@ class BaseInputTransport(FrameProcessor):
await self._handle_bot_interruption(frame) await self._handle_bot_interruption(frame)
elif isinstance(frame, EmulateUserStartedSpeakingFrame): elif isinstance(frame, EmulateUserStartedSpeakingFrame):
logger.debug("Emulating user started speaking") logger.debug("Emulating user started speaking")
await self._handle_user_interruption(UserStartedSpeakingFrame()) await self._handle_user_interruption(UserStartedSpeakingFrame(emulated=True))
elif isinstance(frame, EmulateUserStoppedSpeakingFrame): elif isinstance(frame, EmulateUserStoppedSpeakingFrame):
logger.debug("Emulating user stopped speaking") logger.debug("Emulating user stopped speaking")
await self._handle_user_interruption(UserStoppedSpeakingFrame()) await self._handle_user_interruption(UserStoppedSpeakingFrame(emulated=True))
# All other system frames # All other system frames
elif isinstance(frame, SystemFrame): elif isinstance(frame, SystemFrame):
await self.push_frame(frame, direction) await self.push_frame(frame, direction)