feat: add generation_config support for Cartesia Sonic-3
Add GenerationConfig dataclass with volume, speed, and emotion parameters for Cartesia Sonic-3 TTS models. This enables fine-grained control over speech generation including volume (0.5-2.0), speed (0.6-1.5), and emotion (60+ options). Changes: - Add GenerationConfig dataclass with proper Google-style docstrings - Update CartesiaTTSService.InputParams to include generation_config - Update CartesiaHttpTTSService.InputParams to include generation_config - Modify _build_msg() to include generation_config in WebSocket messages - Modify run_tts() to include generation_config in HTTP requests - Maintain backward compatibility with existing speed and emotion parameters The legacy speed (literal strings) and emotion (list) parameters remain available for non-Sonic-3 models.
This commit is contained in:
@@ -10,6 +10,7 @@ import base64
|
|||||||
import json
|
import json
|
||||||
import uuid
|
import uuid
|
||||||
import warnings
|
import warnings
|
||||||
|
from dataclasses import dataclass
|
||||||
from typing import AsyncGenerator, List, Literal, Optional, Union
|
from typing import AsyncGenerator, List, Literal, Optional, Union
|
||||||
|
|
||||||
from loguru import logger
|
from loguru import logger
|
||||||
@@ -48,6 +49,27 @@ except ModuleNotFoundError as e:
|
|||||||
raise Exception(f"Missing module: {e}")
|
raise Exception(f"Missing module: {e}")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class GenerationConfig:
|
||||||
|
"""Configuration for Cartesia Sonic-3 generation parameters.
|
||||||
|
|
||||||
|
Sonic-3 interprets these parameters as guidance to ensure natural speech.
|
||||||
|
Test against your content for best results.
|
||||||
|
|
||||||
|
Parameters:
|
||||||
|
volume: Volume multiplier for generated speech. Valid range: [0.5, 2.0]. Default is 1.0.
|
||||||
|
speed: Speed multiplier for generated speech. Valid range: [0.6, 1.5]. Default is 1.0.
|
||||||
|
emotion: Single emotion string to guide the emotional tone. Examples include neutral,
|
||||||
|
angry, excited, content, sad, scared. Over 60 emotions are supported. For best
|
||||||
|
results, use with recommended voices: Leo, Jace, Kyle, Gavin, Maya, Tessa, Dana,
|
||||||
|
and Marian.
|
||||||
|
"""
|
||||||
|
|
||||||
|
volume: Optional[float] = None
|
||||||
|
speed: Optional[float] = None
|
||||||
|
emotion: Optional[str] = None
|
||||||
|
|
||||||
|
|
||||||
def language_to_cartesia_language(language: Language) -> Optional[str]:
|
def language_to_cartesia_language(language: Language) -> Optional[str]:
|
||||||
"""Convert a Language enum to Cartesia language code.
|
"""Convert a Language enum to Cartesia language code.
|
||||||
|
|
||||||
@@ -101,16 +123,19 @@ class CartesiaTTSService(AudioContextWordTTSService):
|
|||||||
|
|
||||||
Parameters:
|
Parameters:
|
||||||
language: Language to use for synthesis.
|
language: Language to use for synthesis.
|
||||||
speed: Voice speed control.
|
speed: Voice speed control for non-Sonic-3 models (literal values).
|
||||||
emotion: List of emotion controls.
|
emotion: List of emotion controls for non-Sonic-3 models.
|
||||||
|
|
||||||
.. deprecated:: 0.0.68
|
.. deprecated:: 0.0.68
|
||||||
The `emotion` parameter is deprecated and will be removed in a future version.
|
The `emotion` parameter is deprecated and will be removed in a future version.
|
||||||
|
generation_config: Generation configuration for Sonic-3 models. Includes volume,
|
||||||
|
speed (numeric), and emotion (string) parameters.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
language: Optional[Language] = Language.EN
|
language: Optional[Language] = Language.EN
|
||||||
speed: Optional[Literal["slow", "normal", "fast"]] = None
|
speed: Optional[Literal["slow", "normal", "fast"]] = None
|
||||||
emotion: Optional[List[str]] = []
|
emotion: Optional[List[str]] = []
|
||||||
|
generation_config: Optional[GenerationConfig] = None
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
@@ -179,6 +204,7 @@ class CartesiaTTSService(AudioContextWordTTSService):
|
|||||||
else "en",
|
else "en",
|
||||||
"speed": params.speed,
|
"speed": params.speed,
|
||||||
"emotion": params.emotion,
|
"emotion": params.emotion,
|
||||||
|
"generation_config": params.generation_config,
|
||||||
}
|
}
|
||||||
self.set_model_name(model)
|
self.set_model_name(model)
|
||||||
self.set_voice(voice_id)
|
self.set_voice(voice_id)
|
||||||
@@ -297,6 +323,17 @@ class CartesiaTTSService(AudioContextWordTTSService):
|
|||||||
if self._settings["speed"]:
|
if self._settings["speed"]:
|
||||||
msg["speed"] = self._settings["speed"]
|
msg["speed"] = self._settings["speed"]
|
||||||
|
|
||||||
|
if self._settings["generation_config"]:
|
||||||
|
gen_config = {}
|
||||||
|
if self._settings["generation_config"].volume is not None:
|
||||||
|
gen_config["volume"] = self._settings["generation_config"].volume
|
||||||
|
if self._settings["generation_config"].speed is not None:
|
||||||
|
gen_config["speed"] = self._settings["generation_config"].speed
|
||||||
|
if self._settings["generation_config"].emotion is not None:
|
||||||
|
gen_config["emotion"] = self._settings["generation_config"].emotion
|
||||||
|
if gen_config:
|
||||||
|
msg["generation_config"] = gen_config
|
||||||
|
|
||||||
return json.dumps(msg)
|
return json.dumps(msg)
|
||||||
|
|
||||||
async def start(self, frame: StartFrame):
|
async def start(self, frame: StartFrame):
|
||||||
@@ -482,16 +519,19 @@ class CartesiaHttpTTSService(TTSService):
|
|||||||
|
|
||||||
Parameters:
|
Parameters:
|
||||||
language: Language to use for synthesis.
|
language: Language to use for synthesis.
|
||||||
speed: Voice speed control.
|
speed: Voice speed control for non-Sonic-3 models (literal values).
|
||||||
emotion: List of emotion controls.
|
emotion: List of emotion controls for non-Sonic-3 models.
|
||||||
|
|
||||||
.. deprecated:: 0.0.68
|
.. deprecated:: 0.0.68
|
||||||
The `emotion` parameter is deprecated and will be removed in a future version.
|
The `emotion` parameter is deprecated and will be removed in a future version.
|
||||||
|
generation_config: Generation configuration for Sonic-3 models. Includes volume,
|
||||||
|
speed (numeric), and emotion (string) parameters.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
language: Optional[Language] = Language.EN
|
language: Optional[Language] = Language.EN
|
||||||
speed: Optional[Literal["slow", "normal", "fast"]] = None
|
speed: Optional[Literal["slow", "normal", "fast"]] = None
|
||||||
emotion: Optional[List[str]] = Field(default_factory=list)
|
emotion: Optional[List[str]] = Field(default_factory=list)
|
||||||
|
generation_config: Optional[GenerationConfig] = None
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
@@ -539,6 +579,7 @@ class CartesiaHttpTTSService(TTSService):
|
|||||||
else "en",
|
else "en",
|
||||||
"speed": params.speed,
|
"speed": params.speed,
|
||||||
"emotion": params.emotion,
|
"emotion": params.emotion,
|
||||||
|
"generation_config": params.generation_config,
|
||||||
}
|
}
|
||||||
self.set_voice(voice_id)
|
self.set_voice(voice_id)
|
||||||
self.set_model_name(model)
|
self.set_model_name(model)
|
||||||
@@ -632,6 +673,17 @@ class CartesiaHttpTTSService(TTSService):
|
|||||||
if self._settings["speed"]:
|
if self._settings["speed"]:
|
||||||
payload["speed"] = self._settings["speed"]
|
payload["speed"] = self._settings["speed"]
|
||||||
|
|
||||||
|
if self._settings["generation_config"]:
|
||||||
|
gen_config = {}
|
||||||
|
if self._settings["generation_config"].volume is not None:
|
||||||
|
gen_config["volume"] = self._settings["generation_config"].volume
|
||||||
|
if self._settings["generation_config"].speed is not None:
|
||||||
|
gen_config["speed"] = self._settings["generation_config"].speed
|
||||||
|
if self._settings["generation_config"].emotion is not None:
|
||||||
|
gen_config["emotion"] = self._settings["generation_config"].emotion
|
||||||
|
if gen_config:
|
||||||
|
payload["generation_config"] = gen_config
|
||||||
|
|
||||||
yield TTSStartedFrame()
|
yield TTSStartedFrame()
|
||||||
|
|
||||||
session = await self._client._get_session()
|
session = await self._client._get_session()
|
||||||
|
|||||||
Reference in New Issue
Block a user