Update SpeechmaticsSTTService to use the python voice SDK
This commit is contained in:
15
changelog/3225.changed.md
Normal file
15
changelog/3225.changed.md
Normal file
@@ -0,0 +1,15 @@
|
|||||||
|
- Updated `SpeechmaticsSTTService` to use new Python Voice SDK with improved VAD,
|
||||||
|
Smart Turn capabilities, and brings dramatic improvements to latency without
|
||||||
|
any impact on accuracy. Use the `turn_detection_mode` parameter to control the
|
||||||
|
endpointing of speech, with `TurnDetectionMode.EXTERNAL` (default),
|
||||||
|
`TurnDetectionMode.ADAPTIVE`, or `TurnDetectionMode.SMART_TURN`.
|
||||||
|
```python
|
||||||
|
stt = SpeechmaticsSTTService(
|
||||||
|
api_key=os.getenv("SPEECHMATICS_API_KEY"),
|
||||||
|
params=SpeechmaticsSTTService.InputParams(
|
||||||
|
language=Language.EN,
|
||||||
|
turn_detection_mode=SpeechmaticsSTTService.TurnDetectionMode.ADAPTIVE,
|
||||||
|
speaker_active_format="<{speaker_id}>{text}</{speaker_id}>",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
```
|
||||||
4
changelog/3225.deprecated.md
Normal file
4
changelog/3225.deprecated.md
Normal file
@@ -0,0 +1,4 @@
|
|||||||
|
- For `SpeechmaticsSTTService`, the `end_of_utterance_mode` parameter is deprecated.
|
||||||
|
Use the new `turn_detection_mode` parameter instead, with `TurnDetectionMode.EXTERNAL`,
|
||||||
|
`TurnDetectionMode.ADAPTIVE`, or `TurnDetectionMode.SMART_TURN`. The `enable_vad`
|
||||||
|
parameter is also deprecated and is inferred from the `turn_detection_mode`.
|
||||||
@@ -73,7 +73,7 @@ async def run_bot(transport: BaseTransport, runner_args: RunnerArguments):
|
|||||||
|
|
||||||
4. Text-to-Speech (TTS)
|
4. Text-to-Speech (TTS)
|
||||||
- Low latency streaming audio synthesis
|
- Low latency streaming audio synthesis
|
||||||
- Multiple voice options available including `sarah`, `theo`, and `megan`
|
- Multiple voice options available including `sarah`, `theo`, `megan` and `jack`
|
||||||
|
|
||||||
5. Configuration Options
|
5. Configuration Options
|
||||||
- `operating_point` parameter defaults to `ENHANCED` for optimal accuracy
|
- `operating_point` parameter defaults to `ENHANCED` for optimal accuracy
|
||||||
@@ -92,10 +92,8 @@ async def run_bot(transport: BaseTransport, runner_args: RunnerArguments):
|
|||||||
api_key=os.getenv("SPEECHMATICS_API_KEY"),
|
api_key=os.getenv("SPEECHMATICS_API_KEY"),
|
||||||
params=SpeechmaticsSTTService.InputParams(
|
params=SpeechmaticsSTTService.InputParams(
|
||||||
language=Language.EN,
|
language=Language.EN,
|
||||||
enable_vad=True,
|
turn_detection_mode=SpeechmaticsSTTService.TurnDetectionMode.ADAPTIVE,
|
||||||
enable_diarization=True,
|
# focus_speakers=["S1"],
|
||||||
focus_speakers=["S1"],
|
|
||||||
end_of_utterance_silence_trigger=0.5,
|
|
||||||
speaker_active_format="<{speaker_id}>{text}</{speaker_id}>",
|
speaker_active_format="<{speaker_id}>{text}</{speaker_id}>",
|
||||||
speaker_passive_format="<PASSIVE><{speaker_id}>{text}</{speaker_id}></PASSIVE>",
|
speaker_passive_format="<PASSIVE><{speaker_id}>{text}</{speaker_id}></PASSIVE>",
|
||||||
),
|
),
|
||||||
|
|||||||
@@ -70,7 +70,7 @@ async def run_bot(transport: BaseTransport, runner_args: RunnerArguments):
|
|||||||
|
|
||||||
TTS Features:
|
TTS Features:
|
||||||
- Low latency streaming audio synthesis
|
- Low latency streaming audio synthesis
|
||||||
- Multiple voice options available including `sarah`, `theo`, and `megan`
|
- Multiple voice options available including `sarah`, `theo`, `megan` and `jack`
|
||||||
|
|
||||||
For more information:
|
For more information:
|
||||||
- STT: https://docs.speechmatics.com/rt-api-ref
|
- STT: https://docs.speechmatics.com/rt-api-ref
|
||||||
@@ -83,8 +83,6 @@ async def run_bot(transport: BaseTransport, runner_args: RunnerArguments):
|
|||||||
api_key=os.getenv("SPEECHMATICS_API_KEY"),
|
api_key=os.getenv("SPEECHMATICS_API_KEY"),
|
||||||
params=SpeechmaticsSTTService.InputParams(
|
params=SpeechmaticsSTTService.InputParams(
|
||||||
language=Language.EN,
|
language=Language.EN,
|
||||||
enable_diarization=True,
|
|
||||||
end_of_utterance_silence_trigger=0.5,
|
|
||||||
speaker_active_format="<{speaker_id}>{text}</{speaker_id}>",
|
speaker_active_format="<{speaker_id}>{text}</{speaker_id}>",
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -68,7 +68,6 @@ async def run_bot(transport: BaseTransport, runner_args: RunnerArguments):
|
|||||||
api_key=os.getenv("SPEECHMATICS_API_KEY"),
|
api_key=os.getenv("SPEECHMATICS_API_KEY"),
|
||||||
params=SpeechmaticsSTTService.InputParams(
|
params=SpeechmaticsSTTService.InputParams(
|
||||||
language=Language.EN,
|
language=Language.EN,
|
||||||
enable_diarization=True,
|
|
||||||
speaker_active_format="<{speaker_id}>{text}</{speaker_id}>",
|
speaker_active_format="<{speaker_id}>{text}</{speaker_id}>",
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -105,7 +105,7 @@ silero = [ "onnxruntime>=1.20.1,<2" ]
|
|||||||
simli = [ "simli-ai~=1.0.3"]
|
simli = [ "simli-ai~=1.0.3"]
|
||||||
soniox = [ "pipecat-ai[websockets-base]" ]
|
soniox = [ "pipecat-ai[websockets-base]" ]
|
||||||
soundfile = [ "soundfile~=0.13.1" ]
|
soundfile = [ "soundfile~=0.13.1" ]
|
||||||
speechmatics = [ "speechmatics-rt>=0.5.0" ]
|
speechmatics = [ "speechmatics-voice[smart]>=0.2.4" ]
|
||||||
strands = [ "strands-agents>=1.9.1,<2" ]
|
strands = [ "strands-agents>=1.9.1,<2" ]
|
||||||
tavus=[]
|
tavus=[]
|
||||||
together = []
|
together = []
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user