Merge pull request #1766 from pipecat-ai/aleix/daily-fix-multiple-audio-video-sources
fix multiple audio video sources
This commit is contained in:
@@ -9,6 +9,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
### Added
|
||||
|
||||
- `UserImageRequestFrame.video_source` field has been added to request an image
|
||||
from the desired video source.
|
||||
|
||||
- Added support for the AWS Nova Sonic speech-to-speech model with the new
|
||||
`AWSNovaSonicLLMService`.
|
||||
See https://docs.aws.amazon.com/nova/latest/userguide/speech.html.
|
||||
@@ -39,6 +42,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a `DailyTransport` issue that was causing issues when multiple audio or
|
||||
video sources where being captured.
|
||||
|
||||
- Fixed a `UltravoxSTTService` issue that would cause the service to generate
|
||||
all tokens as one word.
|
||||
|
||||
|
||||
@@ -715,9 +715,10 @@ class UserImageRequestFrame(SystemFrame):
|
||||
context: Optional[Any] = None
|
||||
function_name: Optional[str] = None
|
||||
tool_call_id: Optional[str] = None
|
||||
video_source: Optional[str] = None
|
||||
|
||||
def __str__(self):
|
||||
return f"{self.name}(user: {self.user_id}, function: {self.function_name}, request: {self.tool_call_id})"
|
||||
return f"{self.name}(user: {self.user_id}, video_source: {self.video_source}, function: {self.function_name}, request: {self.tool_call_id})"
|
||||
|
||||
|
||||
@dataclass
|
||||
|
||||
@@ -190,6 +190,7 @@ class LLMService(AIService):
|
||||
function_name: Optional[str] = None,
|
||||
tool_call_id: Optional[str] = None,
|
||||
text_content: Optional[str] = None,
|
||||
video_source: Optional[str] = None,
|
||||
):
|
||||
await self.push_frame(
|
||||
UserImageRequestFrame(
|
||||
@@ -197,6 +198,7 @@ class LLMService(AIService):
|
||||
function_name=function_name,
|
||||
tool_call_id=tool_call_id,
|
||||
context=text_content,
|
||||
video_source=video_source,
|
||||
),
|
||||
FrameDirection.UPSTREAM,
|
||||
)
|
||||
|
||||
@@ -700,7 +700,7 @@ class DailyTransportClient(EventHandler):
|
||||
|
||||
await self.update_subscriptions(participant_settings={participant_id: media})
|
||||
|
||||
self._audio_renderers[participant_id] = {audio_source: callback}
|
||||
self._audio_renderers.setdefault(participant_id, {})[audio_source] = callback
|
||||
|
||||
self._client.set_audio_renderer(
|
||||
participant_id,
|
||||
@@ -724,7 +724,7 @@ class DailyTransportClient(EventHandler):
|
||||
|
||||
await self.update_subscriptions(participant_settings={participant_id: media})
|
||||
|
||||
self._video_renderers[participant_id] = {video_source: callback}
|
||||
self._video_renderers.setdefault(participant_id, {})[video_source] = callback
|
||||
|
||||
self._client.set_video_renderer(
|
||||
participant_id,
|
||||
@@ -1061,12 +1061,13 @@ class DailyInputTransport(BaseInputTransport):
|
||||
video_source: str = "camera",
|
||||
color_format: str = "RGB",
|
||||
):
|
||||
self._video_renderers[participant_id] = {
|
||||
video_source: {
|
||||
"framerate": framerate,
|
||||
"timestamp": 0,
|
||||
"render_next_frame": [],
|
||||
}
|
||||
if participant_id not in self._video_renderers:
|
||||
self._video_renderers[participant_id] = {}
|
||||
|
||||
self._video_renderers[participant_id][video_source] = {
|
||||
"framerate": framerate,
|
||||
"timestamp": 0,
|
||||
"render_next_frame": [],
|
||||
}
|
||||
|
||||
await self._client.capture_participant_video(
|
||||
@@ -1075,7 +1076,8 @@ class DailyInputTransport(BaseInputTransport):
|
||||
|
||||
async def request_participant_image(self, frame: UserImageRequestFrame):
|
||||
if frame.user_id in self._video_renderers:
|
||||
self._video_renderers[frame.user_id]["render_next_frame"].append(frame)
|
||||
video_source = frame.video_source if frame.video_source else "camera"
|
||||
self._video_renderers[frame.user_id][video_source]["render_next_frame"].append(frame)
|
||||
|
||||
async def _on_participant_video_frame(
|
||||
self, participant_id: str, video_frame: VideoFrame, video_source: str
|
||||
|
||||
Reference in New Issue
Block a user