Merge pull request #2635 from pipecat-ai/mb/openai-tts-speed

This commit is contained in:
Mark Backman
2025-09-11 14:22:51 -07:00
committed by GitHub
2 changed files with 19 additions and 8 deletions

View File

@@ -9,6 +9,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added ### Added
- Added a `speed` arg to `OpenAITTSService` to control the speed of the voice
response.
- Added `FrameProcessor.push_interruption_task_frame_and_wait()`. Use this - Added `FrameProcessor.push_interruption_task_frame_and_wait()`. Use this
method to programatically interrupt the bot from any part of the method to programatically interrupt the bot from any part of the
pipeline. This guarantees that all the processors in the pipeline are pipeline. This guarantees that all the processors in the pipeline are

View File

@@ -64,6 +64,7 @@ class OpenAITTSService(TTSService):
model: str = "gpt-4o-mini-tts", model: str = "gpt-4o-mini-tts",
sample_rate: Optional[int] = None, sample_rate: Optional[int] = None,
instructions: Optional[str] = None, instructions: Optional[str] = None,
speed: Optional[float] = None,
**kwargs, **kwargs,
): ):
"""Initialize OpenAI TTS service. """Initialize OpenAI TTS service.
@@ -75,6 +76,7 @@ class OpenAITTSService(TTSService):
model: TTS model to use. Defaults to "gpt-4o-mini-tts". model: TTS model to use. Defaults to "gpt-4o-mini-tts".
sample_rate: Output audio sample rate in Hz. If None, uses OpenAI's default 24kHz. sample_rate: Output audio sample rate in Hz. If None, uses OpenAI's default 24kHz.
instructions: Optional instructions to guide voice synthesis behavior. instructions: Optional instructions to guide voice synthesis behavior.
speed: Voice speed control (0.25 to 4.0, default 1.0).
**kwargs: Additional keyword arguments passed to TTSService. **kwargs: Additional keyword arguments passed to TTSService.
""" """
if sample_rate and sample_rate != self.OPENAI_SAMPLE_RATE: if sample_rate and sample_rate != self.OPENAI_SAMPLE_RATE:
@@ -84,6 +86,7 @@ class OpenAITTSService(TTSService):
) )
super().__init__(sample_rate=sample_rate, **kwargs) super().__init__(sample_rate=sample_rate, **kwargs)
self._speed = speed
self.set_model_name(model) self.set_model_name(model)
self.set_voice(voice) self.set_voice(voice)
self._instructions = instructions self._instructions = instructions
@@ -133,17 +136,22 @@ class OpenAITTSService(TTSService):
try: try:
await self.start_ttfb_metrics() await self.start_ttfb_metrics()
# Setup extra body parameters # Setup API parameters
extra_body = {} create_params = {
"input": text,
"model": self.model_name,
"voice": VALID_VOICES[self._voice_id],
"response_format": "pcm",
}
if self._instructions: if self._instructions:
extra_body["instructions"] = self._instructions create_params["instructions"] = self._instructions
if self._speed:
create_params["speed"] = self._speed
async with self._client.audio.speech.with_streaming_response.create( async with self._client.audio.speech.with_streaming_response.create(
input=text, **create_params
model=self.model_name,
voice=VALID_VOICES[self._voice_id],
response_format="pcm",
extra_body=extra_body,
) as r: ) as r:
if r.status_code != 200: if r.status_code != 200:
error = await r.text() error = await r.text()