Merge branch 'main' into piper-tts
# Conflicts: # CHANGELOG.md
This commit is contained in:
14
CHANGELOG.md
14
CHANGELOG.md
@@ -12,6 +12,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
- Added support for a new TTS service, `PiperTTSService`.
|
- Added support for a new TTS service, `PiperTTSService`.
|
||||||
(see https://github.com/rhasspy/piper/)
|
(see https://github.com/rhasspy/piper/)
|
||||||
|
|
||||||
|
- It is now possible to tell whether `UserStartedSpeakingFrame` or
|
||||||
|
`UserStoppedSpeakingFrame` have been generated because of emulation frames.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- Fixed an issue that would cause `SegmentedSTTService` based services
|
||||||
|
(e.g. `OpenAISTTService`) to try to transcribe non-spoken audio, causing
|
||||||
|
invalid transcriptions.
|
||||||
|
|
||||||
## [0.0.61] - 2025-03-26
|
## [0.0.61] - 2025-03-26
|
||||||
|
|
||||||
### Added
|
### Added
|
||||||
@@ -28,9 +37,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
- ElevenLabs TTS services now support a sample rate of 8000.
|
- ElevenLabs TTS services now support a sample rate of 8000.
|
||||||
|
|
||||||
- Added support for `instructions` in `OpenAITTSService`
|
- Added support for `instructions` in `OpenAITTSService`.
|
||||||
|
|
||||||
- Added support for `base_url` in `OpenAIImageGenService` and `OpenAITTSService`
|
- Added support for `base_url` in `OpenAIImageGenService` and
|
||||||
|
`OpenAITTSService`.
|
||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
|
|||||||
@@ -562,14 +562,14 @@ class UserStartedSpeakingFrame(SystemFrame):
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
pass
|
emulated: bool = False
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class UserStoppedSpeakingFrame(SystemFrame):
|
class UserStoppedSpeakingFrame(SystemFrame):
|
||||||
"""Emitted by the VAD to indicate that a user stopped speaking."""
|
"""Emitted by the VAD to indicate that a user stopped speaking."""
|
||||||
|
|
||||||
pass
|
emulated: bool = False
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
|
|||||||
@@ -249,7 +249,7 @@ class LLMUserContextAggregator(LLMContextResponseAggregator):
|
|||||||
self._waiting_for_aggregation = False
|
self._waiting_for_aggregation = False
|
||||||
|
|
||||||
async def handle_aggregation(self, aggregation: str):
|
async def handle_aggregation(self, aggregation: str):
|
||||||
self._context.add_message({"role": self.role, "content": self._aggregation})
|
self._context.add_message({"role": self.role, "content": aggregation})
|
||||||
|
|
||||||
async def process_frame(self, frame: Frame, direction: FrameDirection):
|
async def process_frame(self, frame: Frame, direction: FrameDirection):
|
||||||
await super().process_frame(frame, direction)
|
await super().process_frame(frame, direction)
|
||||||
@@ -290,12 +290,14 @@ class LLMUserContextAggregator(LLMContextResponseAggregator):
|
|||||||
|
|
||||||
async def push_aggregation(self):
|
async def push_aggregation(self):
|
||||||
if len(self._aggregation) > 0:
|
if len(self._aggregation) > 0:
|
||||||
await self.handle_aggregation(self._aggregation)
|
aggregation = self._aggregation
|
||||||
|
|
||||||
# Reset the aggregation. Reset it before pushing it down, otherwise
|
# Reset the aggregation. Reset it before pushing it down, otherwise
|
||||||
# if the tasks gets cancelled we won't be able to clear things up.
|
# if the tasks gets cancelled we won't be able to clear things up.
|
||||||
self.reset()
|
self.reset()
|
||||||
|
|
||||||
|
await self.handle_aggregation(aggregation)
|
||||||
|
|
||||||
frame = OpenAILLMContextFrame(self._context)
|
frame = OpenAILLMContextFrame(self._context)
|
||||||
await self.push_frame(frame)
|
await self.push_frame(frame)
|
||||||
|
|
||||||
@@ -308,10 +310,16 @@ class LLMUserContextAggregator(LLMContextResponseAggregator):
|
|||||||
async def _cancel(self, frame: CancelFrame):
|
async def _cancel(self, frame: CancelFrame):
|
||||||
await self._cancel_aggregation_task()
|
await self._cancel_aggregation_task()
|
||||||
|
|
||||||
async def _handle_user_started_speaking(self, _: UserStartedSpeakingFrame):
|
async def _handle_user_started_speaking(self, frame: UserStartedSpeakingFrame):
|
||||||
self._user_speaking = True
|
self._user_speaking = True
|
||||||
self._waiting_for_aggregation = True
|
self._waiting_for_aggregation = True
|
||||||
|
|
||||||
|
# If we get a non-emulated UserStartedSpeakingFrame but we are in the
|
||||||
|
# middle of emulating VAD, let's stop emulating VAD (i.e. don't send the
|
||||||
|
# EmulateUserStoppedSpeakingFrame).
|
||||||
|
if not frame.emulated and self._emulating_vad:
|
||||||
|
self._emulating_vad = False
|
||||||
|
|
||||||
async def _handle_user_stopped_speaking(self, _: UserStoppedSpeakingFrame):
|
async def _handle_user_stopped_speaking(self, _: UserStoppedSpeakingFrame):
|
||||||
self._user_speaking = False
|
self._user_speaking = False
|
||||||
# We just stopped speaking. Let's see if there's some aggregation to
|
# We just stopped speaking. Let's see if there's some aggregation to
|
||||||
|
|||||||
@@ -1048,9 +1048,14 @@ class SegmentedSTTService(STTService):
|
|||||||
await self._handle_user_stopped_speaking(frame)
|
await self._handle_user_stopped_speaking(frame)
|
||||||
|
|
||||||
async def _handle_user_started_speaking(self, frame: UserStartedSpeakingFrame):
|
async def _handle_user_started_speaking(self, frame: UserStartedSpeakingFrame):
|
||||||
|
if frame.emulated:
|
||||||
|
return
|
||||||
self._user_speaking = True
|
self._user_speaking = True
|
||||||
|
|
||||||
async def _handle_user_stopped_speaking(self, frame: UserStoppedSpeakingFrame):
|
async def _handle_user_stopped_speaking(self, frame: UserStoppedSpeakingFrame):
|
||||||
|
if frame.emulated:
|
||||||
|
return
|
||||||
|
|
||||||
self._user_speaking = False
|
self._user_speaking = False
|
||||||
|
|
||||||
content = io.BytesIO()
|
content = io.BytesIO()
|
||||||
@@ -1068,7 +1073,7 @@ class SegmentedSTTService(STTService):
|
|||||||
self._audio_buffer.clear()
|
self._audio_buffer.clear()
|
||||||
|
|
||||||
async def process_audio_frame(self, frame: AudioRawFrame, direction: FrameDirection):
|
async def process_audio_frame(self, frame: AudioRawFrame, direction: FrameDirection):
|
||||||
# If the user is speaking the audio buffer will keep growin.
|
# If the user is speaking the audio buffer will keep growing.
|
||||||
self._audio_buffer += frame.audio
|
self._audio_buffer += frame.audio
|
||||||
|
|
||||||
# If the user is not speaking we keep just a little bit of audio.
|
# If the user is not speaking we keep just a little bit of audio.
|
||||||
|
|||||||
@@ -117,10 +117,10 @@ class BaseInputTransport(FrameProcessor):
|
|||||||
await self._handle_bot_interruption(frame)
|
await self._handle_bot_interruption(frame)
|
||||||
elif isinstance(frame, EmulateUserStartedSpeakingFrame):
|
elif isinstance(frame, EmulateUserStartedSpeakingFrame):
|
||||||
logger.debug("Emulating user started speaking")
|
logger.debug("Emulating user started speaking")
|
||||||
await self._handle_user_interruption(UserStartedSpeakingFrame())
|
await self._handle_user_interruption(UserStartedSpeakingFrame(emulated=True))
|
||||||
elif isinstance(frame, EmulateUserStoppedSpeakingFrame):
|
elif isinstance(frame, EmulateUserStoppedSpeakingFrame):
|
||||||
logger.debug("Emulating user stopped speaking")
|
logger.debug("Emulating user stopped speaking")
|
||||||
await self._handle_user_interruption(UserStoppedSpeakingFrame())
|
await self._handle_user_interruption(UserStoppedSpeakingFrame(emulated=True))
|
||||||
# All other system frames
|
# All other system frames
|
||||||
elif isinstance(frame, SystemFrame):
|
elif isinstance(frame, SystemFrame):
|
||||||
await self.push_frame(frame, direction)
|
await self.push_frame(frame, direction)
|
||||||
|
|||||||
Reference in New Issue
Block a user