AzureTTS: work around word ordering issue at 8khz sample rate
This commit is contained in:
@@ -277,6 +277,11 @@ class AzureTTSService(WordTTSService, AzureBaseTTSService):
|
|||||||
self._started = False
|
self._started = False
|
||||||
self._first_chunk = True
|
self._first_chunk = True
|
||||||
self._cumulative_audio_offset: float = 0.0 # Cumulative audio duration in seconds
|
self._cumulative_audio_offset: float = 0.0 # Cumulative audio duration in seconds
|
||||||
|
self._current_sentence_base_offset: float = 0.0 # Base offset for current sentence
|
||||||
|
self._current_sentence_duration: float = 0.0 # Duration from Azure callback
|
||||||
|
self._current_sentence_max_word_offset: float = (
|
||||||
|
0.0 # Max word boundary offset seen in current sentence (for 8kHz workaround)
|
||||||
|
)
|
||||||
self._last_word: Optional[str] = None # Track last word for punctuation merging
|
self._last_word: Optional[str] = None # Track last word for punctuation merging
|
||||||
self._last_timestamp: Optional[float] = None # Track last timestamp
|
self._last_timestamp: Optional[float] = None # Track last timestamp
|
||||||
|
|
||||||
@@ -386,8 +391,14 @@ class AzureTTSService(WordTTSService, AzureBaseTTSService):
|
|||||||
word = evt.text
|
word = evt.text
|
||||||
sentence_relative_seconds = evt.audio_offset / 10_000_000.0
|
sentence_relative_seconds = evt.audio_offset / 10_000_000.0
|
||||||
|
|
||||||
# Add cumulative offset to get absolute timestamp across sentences
|
# Use base offset captured at start of run_tts to avoid race conditions
|
||||||
absolute_seconds = self._cumulative_audio_offset + sentence_relative_seconds
|
# with callbacks from overlapping TTS requests
|
||||||
|
absolute_seconds = self._current_sentence_base_offset + sentence_relative_seconds
|
||||||
|
|
||||||
|
# Track max word offset for accurate cumulative timing
|
||||||
|
# (audio_duration from Azure doesn't always match word boundary offsets at 8kHz)
|
||||||
|
if sentence_relative_seconds > self._current_sentence_max_word_offset:
|
||||||
|
self._current_sentence_max_word_offset = sentence_relative_seconds
|
||||||
|
|
||||||
if not word:
|
if not word:
|
||||||
return
|
return
|
||||||
@@ -492,9 +503,9 @@ class AzureTTSService(WordTTSService, AzureBaseTTSService):
|
|||||||
self._last_word = None
|
self._last_word = None
|
||||||
self._last_timestamp = None
|
self._last_timestamp = None
|
||||||
|
|
||||||
# Update cumulative audio offset for next sentence
|
# Store duration for cumulative offset calculation
|
||||||
if evt.result and evt.result.audio_duration:
|
if evt.result and evt.result.audio_duration:
|
||||||
self._cumulative_audio_offset += evt.result.audio_duration.total_seconds()
|
self._current_sentence_duration = evt.result.audio_duration.total_seconds()
|
||||||
|
|
||||||
self._audio_queue.put_nowait(None) # Signal completion
|
self._audio_queue.put_nowait(None) # Signal completion
|
||||||
|
|
||||||
@@ -530,6 +541,9 @@ class AzureTTSService(WordTTSService, AzureBaseTTSService):
|
|||||||
self._started = False
|
self._started = False
|
||||||
self._first_chunk = True
|
self._first_chunk = True
|
||||||
self._cumulative_audio_offset = 0.0
|
self._cumulative_audio_offset = 0.0
|
||||||
|
self._current_sentence_base_offset = 0.0
|
||||||
|
self._current_sentence_duration = 0.0
|
||||||
|
self._current_sentence_max_word_offset = 0.0
|
||||||
self._last_word = None
|
self._last_word = None
|
||||||
self._last_timestamp = None
|
self._last_timestamp = None
|
||||||
|
|
||||||
@@ -604,6 +618,12 @@ class AzureTTSService(WordTTSService, AzureBaseTTSService):
|
|||||||
self._started = True
|
self._started = True
|
||||||
self._first_chunk = True
|
self._first_chunk = True
|
||||||
|
|
||||||
|
# Capture base offset BEFORE starting synthesis to avoid race conditions
|
||||||
|
# Word boundary callbacks will use this value
|
||||||
|
self._current_sentence_base_offset = self._cumulative_audio_offset
|
||||||
|
self._current_sentence_duration = 0.0
|
||||||
|
self._current_sentence_max_word_offset = 0.0
|
||||||
|
|
||||||
ssml = self._construct_ssml(text)
|
ssml = self._construct_ssml(text)
|
||||||
self._speech_synthesizer.speak_ssml_async(ssml)
|
self._speech_synthesizer.speak_ssml_async(ssml)
|
||||||
await self.start_tts_usage_metrics(text)
|
await self.start_tts_usage_metrics(text)
|
||||||
@@ -627,6 +647,16 @@ class AzureTTSService(WordTTSService, AzureBaseTTSService):
|
|||||||
)
|
)
|
||||||
yield frame
|
yield frame
|
||||||
|
|
||||||
|
# Update cumulative offset for next sentence
|
||||||
|
# At 8kHz, Azure's audio_duration doesn't match word boundary offsets,
|
||||||
|
# so we use max_word_offset as a workaround. At other sample rates,
|
||||||
|
# audio_duration is accurate.
|
||||||
|
# TODO: Remove after Azure fixes word boundary timing at 8kHz
|
||||||
|
if self.sample_rate == 8000:
|
||||||
|
self._cumulative_audio_offset += self._current_sentence_max_word_offset
|
||||||
|
else:
|
||||||
|
self._cumulative_audio_offset += self._current_sentence_duration
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
yield ErrorFrame(error=f"Unknown error occurred: {e}")
|
yield ErrorFrame(error=f"Unknown error occurred: {e}")
|
||||||
yield TTSStoppedFrame()
|
yield TTSStoppedFrame()
|
||||||
|
|||||||
Reference in New Issue
Block a user