Merge pull request #2635 from pipecat-ai/mb/openai-tts-speed
This commit is contained in:
@@ -9,6 +9,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
### Added
|
### Added
|
||||||
|
|
||||||
|
- Added a `speed` arg to `OpenAITTSService` to control the speed of the voice
|
||||||
|
response.
|
||||||
|
|
||||||
- Added `FrameProcessor.push_interruption_task_frame_and_wait()`. Use this
|
- Added `FrameProcessor.push_interruption_task_frame_and_wait()`. Use this
|
||||||
method to programatically interrupt the bot from any part of the
|
method to programatically interrupt the bot from any part of the
|
||||||
pipeline. This guarantees that all the processors in the pipeline are
|
pipeline. This guarantees that all the processors in the pipeline are
|
||||||
|
|||||||
@@ -64,6 +64,7 @@ class OpenAITTSService(TTSService):
|
|||||||
model: str = "gpt-4o-mini-tts",
|
model: str = "gpt-4o-mini-tts",
|
||||||
sample_rate: Optional[int] = None,
|
sample_rate: Optional[int] = None,
|
||||||
instructions: Optional[str] = None,
|
instructions: Optional[str] = None,
|
||||||
|
speed: Optional[float] = None,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
"""Initialize OpenAI TTS service.
|
"""Initialize OpenAI TTS service.
|
||||||
@@ -75,6 +76,7 @@ class OpenAITTSService(TTSService):
|
|||||||
model: TTS model to use. Defaults to "gpt-4o-mini-tts".
|
model: TTS model to use. Defaults to "gpt-4o-mini-tts".
|
||||||
sample_rate: Output audio sample rate in Hz. If None, uses OpenAI's default 24kHz.
|
sample_rate: Output audio sample rate in Hz. If None, uses OpenAI's default 24kHz.
|
||||||
instructions: Optional instructions to guide voice synthesis behavior.
|
instructions: Optional instructions to guide voice synthesis behavior.
|
||||||
|
speed: Voice speed control (0.25 to 4.0, default 1.0).
|
||||||
**kwargs: Additional keyword arguments passed to TTSService.
|
**kwargs: Additional keyword arguments passed to TTSService.
|
||||||
"""
|
"""
|
||||||
if sample_rate and sample_rate != self.OPENAI_SAMPLE_RATE:
|
if sample_rate and sample_rate != self.OPENAI_SAMPLE_RATE:
|
||||||
@@ -84,6 +86,7 @@ class OpenAITTSService(TTSService):
|
|||||||
)
|
)
|
||||||
super().__init__(sample_rate=sample_rate, **kwargs)
|
super().__init__(sample_rate=sample_rate, **kwargs)
|
||||||
|
|
||||||
|
self._speed = speed
|
||||||
self.set_model_name(model)
|
self.set_model_name(model)
|
||||||
self.set_voice(voice)
|
self.set_voice(voice)
|
||||||
self._instructions = instructions
|
self._instructions = instructions
|
||||||
@@ -133,17 +136,22 @@ class OpenAITTSService(TTSService):
|
|||||||
try:
|
try:
|
||||||
await self.start_ttfb_metrics()
|
await self.start_ttfb_metrics()
|
||||||
|
|
||||||
# Setup extra body parameters
|
# Setup API parameters
|
||||||
extra_body = {}
|
create_params = {
|
||||||
|
"input": text,
|
||||||
|
"model": self.model_name,
|
||||||
|
"voice": VALID_VOICES[self._voice_id],
|
||||||
|
"response_format": "pcm",
|
||||||
|
}
|
||||||
|
|
||||||
if self._instructions:
|
if self._instructions:
|
||||||
extra_body["instructions"] = self._instructions
|
create_params["instructions"] = self._instructions
|
||||||
|
|
||||||
|
if self._speed:
|
||||||
|
create_params["speed"] = self._speed
|
||||||
|
|
||||||
async with self._client.audio.speech.with_streaming_response.create(
|
async with self._client.audio.speech.with_streaming_response.create(
|
||||||
input=text,
|
**create_params
|
||||||
model=self.model_name,
|
|
||||||
voice=VALID_VOICES[self._voice_id],
|
|
||||||
response_format="pcm",
|
|
||||||
extra_body=extra_body,
|
|
||||||
) as r:
|
) as r:
|
||||||
if r.status_code != 200:
|
if r.status_code != 200:
|
||||||
error = await r.text()
|
error = await r.text()
|
||||||
|
|||||||
Reference in New Issue
Block a user