TTS services: Add aggregate_sentences arg
This commit is contained in:
@@ -9,6 +9,13 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
### Added
|
### Added
|
||||||
|
|
||||||
|
- Added an `aggregate_sentences` arg in `CartesiaTTSService`,
|
||||||
|
`ElevenLabsTTSService`, `NeuphonicTTSService` and `RimeTTSService`, where the
|
||||||
|
default value is True. When `aggregate_sentences` is True, the `TTSService`
|
||||||
|
aggregates the LLM streamed tokens into sentences by default. Note: setting
|
||||||
|
the value to False requires a custom processor before the `TTSService` to
|
||||||
|
aggregate LLM tokens.
|
||||||
|
|
||||||
- Added call hang-up error handling in `TwilioFrameSerializer`, which handles
|
- Added call hang-up error handling in `TwilioFrameSerializer`, which handles
|
||||||
the case where the user has hung up before the `TwilioFrameSerializer` hangs
|
the case where the user has hung up before the `TwilioFrameSerializer` hangs
|
||||||
up the call.
|
up the call.
|
||||||
|
|||||||
@@ -121,6 +121,7 @@ class CartesiaTTSService(AudioContextWordTTSService):
|
|||||||
container: str = "raw",
|
container: str = "raw",
|
||||||
params: Optional[InputParams] = None,
|
params: Optional[InputParams] = None,
|
||||||
text_aggregator: Optional[BaseTextAggregator] = None,
|
text_aggregator: Optional[BaseTextAggregator] = None,
|
||||||
|
aggregate_sentences: Optional[bool] = True,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
"""Initialize the Cartesia TTS service.
|
"""Initialize the Cartesia TTS service.
|
||||||
@@ -136,6 +137,7 @@ class CartesiaTTSService(AudioContextWordTTSService):
|
|||||||
container: Audio container format.
|
container: Audio container format.
|
||||||
params: Additional input parameters for voice customization.
|
params: Additional input parameters for voice customization.
|
||||||
text_aggregator: Custom text aggregator for processing input text.
|
text_aggregator: Custom text aggregator for processing input text.
|
||||||
|
aggregate_sentences: Whether to aggregate sentences within the TTSService.
|
||||||
**kwargs: Additional arguments passed to the parent service.
|
**kwargs: Additional arguments passed to the parent service.
|
||||||
"""
|
"""
|
||||||
# Aggregating sentences still gives cleaner-sounding results and fewer
|
# Aggregating sentences still gives cleaner-sounding results and fewer
|
||||||
@@ -149,7 +151,7 @@ class CartesiaTTSService(AudioContextWordTTSService):
|
|||||||
# can use those to generate text frames ourselves aligned with the
|
# can use those to generate text frames ourselves aligned with the
|
||||||
# playout timing of the audio!
|
# playout timing of the audio!
|
||||||
super().__init__(
|
super().__init__(
|
||||||
aggregate_sentences=True,
|
aggregate_sentences=aggregate_sentences,
|
||||||
push_text_frames=False,
|
push_text_frames=False,
|
||||||
pause_frame_processing=True,
|
pause_frame_processing=True,
|
||||||
sample_rate=sample_rate,
|
sample_rate=sample_rate,
|
||||||
|
|||||||
@@ -238,6 +238,7 @@ class ElevenLabsTTSService(AudioContextWordTTSService):
|
|||||||
url: str = "wss://api.elevenlabs.io",
|
url: str = "wss://api.elevenlabs.io",
|
||||||
sample_rate: Optional[int] = None,
|
sample_rate: Optional[int] = None,
|
||||||
params: Optional[InputParams] = None,
|
params: Optional[InputParams] = None,
|
||||||
|
aggregate_sentences: Optional[bool] = True,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
"""Initialize the ElevenLabs TTS service.
|
"""Initialize the ElevenLabs TTS service.
|
||||||
@@ -249,6 +250,7 @@ class ElevenLabsTTSService(AudioContextWordTTSService):
|
|||||||
url: WebSocket URL for ElevenLabs TTS API.
|
url: WebSocket URL for ElevenLabs TTS API.
|
||||||
sample_rate: Audio sample rate. If None, uses default.
|
sample_rate: Audio sample rate. If None, uses default.
|
||||||
params: Additional input parameters for voice customization.
|
params: Additional input parameters for voice customization.
|
||||||
|
aggregate_sentences: Whether to aggregate sentences within the TTSService.
|
||||||
**kwargs: Additional arguments passed to the parent service.
|
**kwargs: Additional arguments passed to the parent service.
|
||||||
"""
|
"""
|
||||||
# Aggregating sentences still gives cleaner-sounding results and fewer
|
# Aggregating sentences still gives cleaner-sounding results and fewer
|
||||||
@@ -266,7 +268,7 @@ class ElevenLabsTTSService(AudioContextWordTTSService):
|
|||||||
# speaking for a while, so we want the parent class to send TTSStopFrame
|
# speaking for a while, so we want the parent class to send TTSStopFrame
|
||||||
# after a short period not receiving any audio.
|
# after a short period not receiving any audio.
|
||||||
super().__init__(
|
super().__init__(
|
||||||
aggregate_sentences=True,
|
aggregate_sentences=aggregate_sentences,
|
||||||
push_text_frames=False,
|
push_text_frames=False,
|
||||||
push_stop_frames=True,
|
push_stop_frames=True,
|
||||||
pause_frame_processing=True,
|
pause_frame_processing=True,
|
||||||
|
|||||||
@@ -110,6 +110,7 @@ class NeuphonicTTSService(InterruptibleTTSService):
|
|||||||
sample_rate: Optional[int] = 22050,
|
sample_rate: Optional[int] = 22050,
|
||||||
encoding: str = "pcm_linear",
|
encoding: str = "pcm_linear",
|
||||||
params: Optional[InputParams] = None,
|
params: Optional[InputParams] = None,
|
||||||
|
aggregate_sentences: Optional[bool] = True,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
"""Initialize the Neuphonic TTS service.
|
"""Initialize the Neuphonic TTS service.
|
||||||
@@ -121,10 +122,11 @@ class NeuphonicTTSService(InterruptibleTTSService):
|
|||||||
sample_rate: Audio sample rate in Hz. Defaults to 22050.
|
sample_rate: Audio sample rate in Hz. Defaults to 22050.
|
||||||
encoding: Audio encoding format. Defaults to "pcm_linear".
|
encoding: Audio encoding format. Defaults to "pcm_linear".
|
||||||
params: Additional input parameters for TTS configuration.
|
params: Additional input parameters for TTS configuration.
|
||||||
|
aggregate_sentences: Whether to aggregate sentences within the TTSService.
|
||||||
**kwargs: Additional arguments passed to parent InterruptibleTTSService.
|
**kwargs: Additional arguments passed to parent InterruptibleTTSService.
|
||||||
"""
|
"""
|
||||||
super().__init__(
|
super().__init__(
|
||||||
aggregate_sentences=True,
|
aggregate_sentences=aggregate_sentences,
|
||||||
push_text_frames=False,
|
push_text_frames=False,
|
||||||
push_stop_frames=True,
|
push_stop_frames=True,
|
||||||
stop_frame_timeout_s=2.0,
|
stop_frame_timeout_s=2.0,
|
||||||
|
|||||||
@@ -99,6 +99,7 @@ class RimeTTSService(AudioContextWordTTSService):
|
|||||||
sample_rate: Optional[int] = None,
|
sample_rate: Optional[int] = None,
|
||||||
params: Optional[InputParams] = None,
|
params: Optional[InputParams] = None,
|
||||||
text_aggregator: Optional[BaseTextAggregator] = None,
|
text_aggregator: Optional[BaseTextAggregator] = None,
|
||||||
|
aggregate_sentences: Optional[bool] = True,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
"""Initialize Rime TTS service.
|
"""Initialize Rime TTS service.
|
||||||
@@ -111,11 +112,12 @@ class RimeTTSService(AudioContextWordTTSService):
|
|||||||
sample_rate: Audio sample rate in Hz.
|
sample_rate: Audio sample rate in Hz.
|
||||||
params: Additional configuration parameters.
|
params: Additional configuration parameters.
|
||||||
text_aggregator: Custom text aggregator for processing input text.
|
text_aggregator: Custom text aggregator for processing input text.
|
||||||
|
aggregate_sentences: Whether to aggregate sentences within the TTSService.
|
||||||
**kwargs: Additional arguments passed to parent class.
|
**kwargs: Additional arguments passed to parent class.
|
||||||
"""
|
"""
|
||||||
# Initialize with parent class settings for proper frame handling
|
# Initialize with parent class settings for proper frame handling
|
||||||
super().__init__(
|
super().__init__(
|
||||||
aggregate_sentences=True,
|
aggregate_sentences=aggregate_sentences,
|
||||||
push_text_frames=False,
|
push_text_frames=False,
|
||||||
push_stop_frames=True,
|
push_stop_frames=True,
|
||||||
pause_frame_processing=True,
|
pause_frame_processing=True,
|
||||||
|
|||||||
Reference in New Issue
Block a user