Merge branch 'main' into cb/patient-intake

# Conflicts:
#	src/dailyai/queue_aggregators.py
#	src/dailyai/queue_frame.py
#	src/dailyai/services/base_transport_service.py
#	src/dailyai/services/daily_transport_service.py
#	src/examples/foundational/06-listen-and-respond.py
#	src/examples/foundational/07-interruptible.py
This commit is contained in:
Chad Bailey
2024-02-28 15:25:03 -06:00
29 changed files with 476 additions and 177 deletions

24
LICENSE Normal file
View File

@@ -0,0 +1,24 @@
BSD 2-Clause License
Copyright (c) 2024, Daily
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the following conditions are met:
1. Redistributions of source code must retain the above copyright notice, this
list of conditions and the following disclaimer.
2. Redistributions in binary form must reproduce the above copyright notice,
this list of conditions and the following disclaimer in the documentation
and/or other materials provided with the distribution.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.

View File

@@ -2,7 +2,7 @@
Build conversational, multi-modal AI apps with real-time voice and video, like this: Build conversational, multi-modal AI apps with real-time voice and video, like this:
_Demo Video_ _Demo Video to come_
With built-in support for many of the best AI platforms (or [add your own](/docs)): With built-in support for many of the best AI platforms (or [add your own](/docs)):
@@ -15,22 +15,6 @@ With built-in support for many of the best AI platforms (or [add your own](/docs
## Step 1: Get Started ## Step 1: Get Started
Installation here. Also sign up for a Daily account, I guess? also we need an ENV
Requires python 3.11 or later. Don't forget virtualenv
pip install vs download and build?
## Step 2: Build Things
Once you've got the SDK working, head over to the [docs folder](/docs) to start building!
---
# Old Readme
This SDK can help you build applications that participate in WebRTC meetings and use various AI services to interact with other participants.
## Build/Install ## Build/Install
_Note that you may need to set up a virtual environment before following the instructions below. For instance, you might need to run the following from the root of the repo:_ _Note that you may need to set up a virtual environment before following the instructions below. For instance, you might need to run the following from the root of the repo:_
@@ -66,23 +50,6 @@ Tou can run the simple sample like so:
``` ```
python src/examples/theoretical-to-real/01-say-one-thing.py -u <url of your Daily meeting> -k <your Daily API Key> python src/examples/theoretical-to-real/01-say-one-thing.py -u <url of your Daily meeting> -k <your Daily API Key>
``` ```
Note that the sample uses Azure's TTS and LLM services. You'll need to set the following environment variables for the sample to work:
```
AZURE_SPEECH_SERVICE_KEY
AZURE_SPEECH_SERVICE_REGION
AZURE_CHATGPT_KEY
AZURE_CHATGPT_ENDPOINT
AZURE_CHATGPT_DEPLOYMENT_ID
```
If you have those environment variables stored in an .env file, you can quickly load them into your terminal's environment by running this:
```bash
export $(grep -v '^#' .env | xargs)
```
## Overview ## Overview
The Daily AI SDK allows you to build applications that can participate in WebRTC sessions and interact with AI Services. Some examples of what you can build with this: The Daily AI SDK allows you to build applications that can participate in WebRTC sessions and interact with AI Services. Some examples of what you can build with this:

View File

@@ -13,10 +13,13 @@ dependencies = [
"fal", "fal",
"faster_whisper", "faster_whisper",
"google-cloud-texttospeech", "google-cloud-texttospeech",
"numpy",
"openai", "openai",
"Pillow", "Pillow",
"pyht", "pyht",
"python-dotenv", "python-dotenv",
"torch",
"pyaudio",
"typing-extensions" "typing-extensions"
] ]

View File

@@ -61,7 +61,8 @@ class LLMContextAggregator(AIService):
# TODO: split up transcription by participant # TODO: split up transcription by participant
if self.complete_sentences: if self.complete_sentences:
# type: ignore -- the linter thinks this isn't a TextQueueFrame, even though we check it above # type: ignore -- the linter thinks this isn't a TextQueueFrame, even
# though we check it above
self.sentence += frame.text self.sentence += frame.text
if self.sentence.endswith((".", "?", "!")): if self.sentence.endswith((".", "?", "!")):
self.messages.append( self.messages.append(
@@ -71,10 +72,9 @@ class LLMContextAggregator(AIService):
print(f"{message['role']}: {message['content']}") print(f"{message['role']}: {message['content']}")
yield LLMMessagesQueueFrame(self.messages) yield LLMMessagesQueueFrame(self.messages)
else: else:
# type: ignore -- the linter thinks this isn't a TextQueueFrame, even though we check it above # type: ignore -- the linter thinks this isn't a TextQueueFrame, even
# though we check it above
self.messages.append({"role": self.role, "content": frame.text}) self.messages.append({"role": self.role, "content": frame.text})
for message in self.messages:
print(f"{message['role']}: {message['content']}")
yield LLMMessagesQueueFrame(self.messages) yield LLMMessagesQueueFrame(self.messages)
async def finalize(self) -> AsyncGenerator[QueueFrame, None]: async def finalize(self) -> AsyncGenerator[QueueFrame, None]:
@@ -91,8 +91,7 @@ class LLMUserContextAggregator(LLMContextAggregator):
messages: list[dict], messages: list[dict],
bot_participant_id=None, bot_participant_id=None,
complete_sentences=True): complete_sentences=True):
super().__init__(messages, "user", bot_participant_id, super().__init__(messages, "user", bot_participant_id, complete_sentences, pass_through=False)
complete_sentences, pass_through=False)
class LLMAssistantContextAggregator(LLMContextAggregator): class LLMAssistantContextAggregator(LLMContextAggregator):

View File

@@ -69,3 +69,9 @@ class LLMMessagesQueueFrame(QueueFrame):
class AppMessageQueueFrame(QueueFrame): class AppMessageQueueFrame(QueueFrame):
message: Any message: Any
participantId: str participantId: str
class UserStartedSpeakingFrame(QueueFrame):
pass
class UserStoppedSpeakingFrame(QueueFrame):
pass

View File

@@ -2,10 +2,15 @@ from abc import abstractmethod
import asyncio import asyncio
import itertools import itertools
import logging import logging
import numpy as np
import pyaudio
import torch
import torchaudio
import queue import queue
import threading import threading
import time import time
from typing import AsyncGenerator from typing import AsyncGenerator
from enum import Enum
from dailyai.queue_frame import ( from dailyai.queue_frame import (
AudioQueueFrame, AudioQueueFrame,
@@ -15,9 +20,58 @@ from dailyai.queue_frame import (
QueueFrame, QueueFrame,
SpriteQueueFrame, SpriteQueueFrame,
StartStreamQueueFrame, StartStreamQueueFrame,
UserStartedSpeakingFrame,
UserStoppedSpeakingFrame
) )
torch.set_num_threads(1)
model, utils = torch.hub.load(repo_or_dir='snakers4/silero-vad',
model='silero_vad',
force_reload=False)
(get_speech_timestamps,
save_audio,
read_audio,
VADIterator,
collect_chunks) = utils
# Taken from utils_vad.py
def validate(model,
inputs: torch.Tensor):
with torch.no_grad():
outs = model(inputs)
return outs
# Provided by Alexander Veysov
def int2float(sound):
abs_max = np.abs(sound).max()
sound = sound.astype('float32')
if abs_max > 0:
sound *= 1/32768
sound = sound.squeeze() # depends on the use case
return sound
FORMAT = pyaudio.paInt16
CHANNELS = 1
SAMPLE_RATE = 16000
CHUNK = int(SAMPLE_RATE / 10)
audio = pyaudio.PyAudio()
class VADState(Enum):
QUIET = 1
STARTING = 2
SPEAKING = 3
STOPPING = 4
class BaseTransportService(): class BaseTransportService():
def __init__( def __init__(
@@ -32,6 +86,22 @@ class BaseTransportService():
self._speaker_enabled = kwargs.get("speaker_enabled") or False self._speaker_enabled = kwargs.get("speaker_enabled") or False
self._speaker_sample_rate = kwargs.get("speaker_sample_rate") or 16000 self._speaker_sample_rate = kwargs.get("speaker_sample_rate") or 16000
self._fps = kwargs.get("fps") or 8 self._fps = kwargs.get("fps") or 8
self._vad_start_s = kwargs.get("vad_start_s") or 0.2
self._vad_stop_s = kwargs.get("vad_stop_s") or 0.8
self._context = kwargs.get("context") or []
self._vad_enabled = kwargs.get("vad_enabled") or False
if self._vad_enabled and self._speaker_enabled:
raise Exception("Sorry, you can't use speaker_enabled and vad_enabled at the same time. Please set one to False.")
self._vad_samples = 1536
vad_frame_s = self._vad_samples / SAMPLE_RATE
self._vad_start_frames = round(self._vad_start_s / vad_frame_s)
self._vad_stop_frames = round(self._vad_stop_s / vad_frame_s)
self._vad_starting_count = 0
self._vad_stopping_count = 0
self._vad_state = VADState.QUIET
self._user_is_speaking = False
duration_minutes = kwargs.get("duration_minutes") or 10 duration_minutes = kwargs.get("duration_minutes") or 10
self._expiration = time.time() + duration_minutes * 60 self._expiration = time.time() + duration_minutes * 60
@@ -72,6 +142,10 @@ class BaseTransportService():
target=self._receive_audio, daemon=True) target=self._receive_audio, daemon=True)
self._receive_audio_thread.start() self._receive_audio_thread.start()
if self._vad_enabled:
self._vad_thread = threading.Thread(target=self._vad, daemon=True)
self._vad_thread.start()
try: try:
while ( while (
time.time() < self._expiration time.time() < self._expiration
@@ -81,6 +155,10 @@ class BaseTransportService():
except Exception as e: except Exception as e:
self._logger.error(f"Exception {e}") self._logger.error(f"Exception {e}")
raise e raise e
finally:
# Do anything that must be done to clean up
self._post_run()
self._stop_threads.set() self._stop_threads.set()
await self.send_queue.put(EndStreamQueueFrame()) await self.send_queue.put(EndStreamQueueFrame())
@@ -91,6 +169,15 @@ class BaseTransportService():
if self._speaker_enabled: if self._speaker_enabled:
self._receive_audio_thread.join() self._receive_audio_thread.join()
if self._vad_enabled:
self._vad_thread.join()
def _post_run(self):
# Note that this function must be idempotent! It can be called multiple times
# if, for example, a keyboard interrupt occurs.
pass
def stop(self): def stop(self):
self._stop_threads.set() self._stop_threads.set()
@@ -118,6 +205,56 @@ class BaseTransportService():
def _prerun(self): def _prerun(self):
pass pass
def _vad(self):
# CB: Starting silero VAD stuff
# TODO-CB: Probably need to force virtual speaker creation if we're
# going to build this in?
# TODO-CB: pyaudio installation
while not self._stop_threads.is_set():
audio_chunk = self.read_audio_frames(self._vad_samples)
audio_int16 = np.frombuffer(audio_chunk, np.int16)
audio_float32 = int2float(audio_int16)
new_confidence = model(
torch.from_numpy(audio_float32), 16000).item()
speaking = new_confidence > 0.5
if speaking:
match self._vad_state:
case VADState.QUIET:
self._vad_state = VADState.STARTING
self._vad_starting_count = 1
case VADState.STARTING:
self._vad_starting_count += 1
case VADState.STOPPING:
self._vad_state = VADState.SPEAKING
self._vad_stopping_count = 0
else:
match self._vad_state:
case VADState.STARTING:
self._vad_state = VADState.QUIET
self._vad_starting_count = 0
case VADState.SPEAKING:
self._vad_state = VADState.STOPPING
self._vad_stopping_count = 1
case VADState.STOPPING:
self._vad_stopping_count += 1
if self._vad_state == VADState.STARTING and self._vad_starting_count >= self._vad_start_frames:
asyncio.run_coroutine_threadsafe(
self.receive_queue.put(
UserStartedSpeakingFrame()), self._loop
)
# self.interrupt()
self._vad_state = VADState.SPEAKING
self._vad_starting_count = 0
if self._vad_state == VADState.STOPPING and self._vad_stopping_count >= self._vad_stop_frames:
asyncio.run_coroutine_threadsafe(
self.receive_queue.put(
UserStoppedSpeakingFrame()), self._loop
)
self._vad_state = VADState.QUIET
self._vad_stopping_count = 0
async def _marshal_frames(self): async def _marshal_frames(self):
while True: while True:
frame: QueueFrame | list = await self.send_queue.get() frame: QueueFrame | list = await self.send_queue.get()

View File

@@ -1,8 +1,8 @@
import asyncio import asyncio
import inspect import inspect
import logging import logging
import signal
import threading import threading
import time
import types import types
from functools import partial from functools import partial
@@ -11,7 +11,7 @@ from dailyai.queue_frame import (
TranscriptionQueueFrame, TranscriptionQueueFrame,
) )
from threading import Thread, Event from threading import Event
from daily import ( from daily import (
EventHandler, EventHandler,
@@ -31,6 +31,7 @@ class DailyTransportService(BaseTransportService, EventHandler):
_speaker_enabled: bool _speaker_enabled: bool
_speaker_sample_rate: int _speaker_sample_rate: int
_vad_enabled: bool
# This is necessary to override EventHandler's __new__ method. # This is necessary to override EventHandler's __new__ method.
def __new__(cls, *args, **kwargs): def __new__(cls, *args, **kwargs):
@@ -45,8 +46,7 @@ class DailyTransportService(BaseTransportService, EventHandler):
start_transcription: bool = False, start_transcription: bool = False,
**kwargs, **kwargs,
): ):
# This will call BaseTransportService.__init__ method, not EventHandler super().__init__(**kwargs) # This will call BaseTransportService.__init__ method, not EventHandler
super().__init__(**kwargs)
self._room_url: str = room_url self._room_url: str = room_url
self._bot_name: str = bot_name self._bot_name: str = bot_name
@@ -146,59 +146,76 @@ class DailyTransportService(BaseTransportService, EventHandler):
"camera", width=self._camera_width, height=self._camera_height, color_format="RGB" "camera", width=self._camera_width, height=self._camera_height, color_format="RGB"
) )
if self._speaker_enabled: if self._speaker_enabled or self._vad_enabled:
self._speaker: VirtualSpeakerDevice = Daily.create_speaker_device( self._speaker: VirtualSpeakerDevice = Daily.create_speaker_device(
"speaker", sample_rate=self._speaker_sample_rate, channels=1 "speaker", sample_rate=self._speaker_sample_rate, channels=1
) )
Daily.select_speaker_device("speaker") Daily.select_speaker_device("speaker")
self.client.set_user_name(self._bot_name) self.client.set_user_name(self._bot_name)
self.client.join(self._room_url, self._token,
completion=self.call_joined) self.client.join(
self._room_url,
self._token,
completion=self.call_joined,
client_settings={
"inputs": {
"camera": {
"isEnabled": True,
"settings": {
"deviceId": "camera",
},
},
"microphone": {
"isEnabled": True,
"settings": {
"deviceId": "mic",
"customConstraints": {
"autoGainControl": {"exact": False},
"echoCancellation": {"exact": False},
"noiseSuppression": {"exact": False},
},
},
},
},
"publishing": {
"camera": {
"sendSettings": {
"maxQuality": "low",
"encodings": {
"low": {
"maxBitrate": 250000,
"scaleResolutionDownBy": 1.333,
"maxFramerate": 8,
}
},
}
}
},
},
)
self._my_participant_id = self.client.participants()["local"]["id"] self._my_participant_id = self.client.participants()["local"]["id"]
self.client.update_inputs( self.client.update_subscription_profiles({
{ "base": {
"camera": { "camera": "unsubscribed",
"isEnabled": True,
"settings": {
"deviceId": "camera",
},
},
"microphone": {
"isEnabled": True,
"settings": {
"deviceId": "mic",
"customConstraints": {
"autoGainControl": {"exact": False},
"echoCancellation": {"exact": False},
"noiseSuppression": {"exact": False},
},
},
},
} }
) })
self.client.update_publishing(
{
"camera": {
"sendSettings": {
"maxQuality": "low",
"encodings": {
"low": {
"maxBitrate": 250000,
"scaleResolutionDownBy": 1.333,
"maxFramerate": 8,
}
},
}
}
}
)
if self._token and self._start_transcription: if self._token and self._start_transcription:
self.client.start_transcription(self.transcription_settings) self.client.start_transcription(self.transcription_settings)
self.original_sigint_handler = signal.getsignal(signal.SIGINT)
signal.signal(signal.SIGINT, self.process_interrupt_handler)
def process_interrupt_handler(self, signum, frame):
self._post_run()
if callable(self.original_sigint_handler):
self.original_sigint_handler(signum, frame)
def _post_run(self):
self.client.leave()
def on_first_other_participant_joined(self): def on_first_other_participant_joined(self):
pass pass

View File

@@ -0,0 +1,36 @@
import os
import aiohttp
import requests
from dailyai.services.ai_services import TTSService
class DeepgramAIService(TTSService):
def __init__(
self,
*,
aiohttp_session: aiohttp.ClientSession,
api_key,
voice,
sample_rate=16000
):
super().__init__()
self._api_key = api_key
self._voice = voice
self._sample_rate = sample_rate
self._aiohttp_session = aiohttp_session
async def run_tts(self, sentence):
self.logger.info(f"Running deepgram tts for {sentence}")
base_url = "https://api.beta.deepgram.com/v1/speak"
request_url = f"{base_url}?model={self._voice}&encoding=linear16&container=none&sample_rate={self._sample_rate}"
headers = {"authorization": f"token {self._api_key}", "Content-Type": "application/json"}
data = {"text": sentence}
async with self._aiohttp_session.post(
request_url, headers=headers, json=data
) as r:
async for chunk in r.content:
if chunk:
yield chunk

View File

@@ -13,7 +13,13 @@ from dailyai.services.ai_services import ImageGenService
class FalImageGenService(ImageGenService): class FalImageGenService(ImageGenService):
def __init__(self, *, image_size, aiohttp_session: aiohttp.ClientSession, key_id=None, key_secret=None): def __init__(
self,
*,
image_size,
aiohttp_session: aiohttp.ClientSession,
key_id=None,
key_secret=None):
super().__init__(image_size) super().__init__(image_size)
self._aiohttp_session = aiohttp_session self._aiohttp_session = aiohttp_session
if key_id: if key_id:

View File

@@ -22,11 +22,15 @@ class LocalTransportService(BaseTransportService):
async def _write_frame_to_tkinter(self, frame: bytes): async def _write_frame_to_tkinter(self, frame: bytes):
data = f"P6 {self._camera_width} {self._camera_height} 255 ".encode() + frame data = f"P6 {self._camera_width} {self._camera_height} 255 ".encode() + frame
photo = tk.PhotoImage(width=self._camera_width, height=self._camera_height, data=data, format="PPM") photo = tk.PhotoImage(
width=self._camera_width,
height=self._camera_height,
data=data,
format="PPM")
self._image_label.config(image=photo) self._image_label.config(image=photo)
# This holds a reference to the photo, preventing it from being garbage collected. # This holds a reference to the photo, preventing it from being garbage collected.
self._image_label.image = photo # type: ignore self._image_label.image = photo # type: ignore
def write_frame_to_camera(self, frame: bytes): def write_frame_to_camera(self, frame: bytes):
if self._camera_enabled and self._loop: if self._camera_enabled and self._loop:

View File

@@ -0,0 +1,42 @@
from openai import AsyncOpenAI
import json
from collections.abc import AsyncGenerator
from dailyai.services.ai_services import LLMService
class OLLamaLLMService(LLMService):
def __init__(self, model="llama2", base_url='http://localhost:11434/v1'):
super().__init__()
self._model = model
self._client = AsyncOpenAI(api_key="ollama", base_url=base_url)
async def get_response(self, messages, stream):
return await self._client.chat.completions.create(
stream=stream,
messages=messages,
model=self._model
)
async def run_llm_async(self, messages) -> AsyncGenerator[str, None]:
messages_for_log = json.dumps(messages)
self.logger.debug(f"Generating chat via openai: {messages_for_log}")
chunks = await self._client.chat.completions.create(model=self._model, stream=True, messages=messages)
async for chunk in chunks:
if len(chunk.choices) == 0:
continue
if chunk.choices[0].delta.content:
yield chunk.choices[0].delta.content
async def run_llm(self, messages) -> str | None:
messages_for_log = json.dumps(messages)
self.logger.debug(f"Generating chat via openai: {messages_for_log}")
response = await self._client.chat.completions.create(model=self._model, stream=False, messages=messages)
if response and len(response.choices) > 0:
return response.choices[0].message.content
else:
return None

View File

@@ -1,15 +1,12 @@
import requests
import aiohttp import aiohttp
import asyncio
from PIL import Image from PIL import Image
import io import io
from openai import AsyncOpenAI from openai import AsyncOpenAI
import os
import json import json
from collections.abc import AsyncGenerator from collections.abc import AsyncGenerator
from dailyai.services.ai_services import AIService, TTSService, LLMService, ImageGenService from dailyai.services.ai_services import LLMService, ImageGenService
class OpenAILLMService(LLMService): class OpenAILLMService(LLMService):

View File

@@ -1,36 +1,40 @@
import io import io
import os
import struct import struct
from pyht import Client from pyht import Client
from dotenv import load_dotenv
from pyht.client import TTSOptions from pyht.client import TTSOptions
from pyht.protos.api_pb2 import Format from pyht.protos.api_pb2 import Format
from services.ai_service import AIService from dailyai.services.ai_services import TTSService
class PlayHTAIService(AIService): class PlayHTAIService(TTSService):
def __init__(self, **kwargs):
super().__init__(**kwargs)
self.speech_key = os.getenv("PLAY_HT_KEY") or '' def __init__(
self.user_id = os.getenv("PLAY_HT_USER_ID") or '' self,
*,
api_key,
user_id,
voice_url
):
super().__init__()
self.speech_key = api_key
self.user_id = user_id
self.client = Client( self.client = Client(
user_id=self.user_id, user_id=self.user_id,
api_key=self.speech_key, api_key=self.speech_key,
) )
self.options = TTSOptions( self.options = TTSOptions(
voice="s3://voice-cloning-zero-shot/820da3d2-3a3b-42e7-844d-e68db835a206/sarah/manifest.json", voice=voice_url,
sample_rate=16000, sample_rate=16000,
quality="higher", quality="higher",
format=Format.FORMAT_WAV) format=Format.FORMAT_WAV)
def close(self): def __del__(self):
super().close()
self.client.close() self.client.close()
def run_tts(self, sentence): async def run_tts(self, sentence):
b = bytearray() b = bytearray()
in_header = True in_header = True
for chunk in self.client.tts(sentence, self.options): for chunk in self.client.tts(sentence, self.options):

View File

@@ -1,29 +0,0 @@
import os
import requests
from services.ai_service import AIService
from PIL import Image
class DeepgramAIService(AIService):
def __init__(self, **kwargs):
super().__init__(**kwargs)
self.api_key = os.getenv("DEEPGRAM_API_KEY")
def get_mic_sample_rate(self):
return 24000
def run_tts(self, sentence):
self.logger.info(f"Running deepgram tts for {sentence}")
base_url = "https://api.beta.deepgram.com/v1/speak"
# move this to an environment variable
voice = os.getenv("DEEPGRAM_VOICE") or "alpha-apollo-en-v1"
request_url = f"{base_url}?model={voice}&encoding=linear16&container=none"
headers = {"authorization": f"token {self.api_key}"}
r = requests.post(request_url, headers=headers, data=sentence)
self.logger.info(
f"audio fetch status code: {r.status_code}, content length: {len(r.content)}"
)
yield r.content

View File

@@ -4,9 +4,11 @@ import os
from dailyai.services.daily_transport_service import DailyTransportService from dailyai.services.daily_transport_service import DailyTransportService
from dailyai.services.elevenlabs_ai_service import ElevenLabsTTSService from dailyai.services.elevenlabs_ai_service import ElevenLabsTTSService
from dailyai.services.playht_ai_service import PlayHTAIService
from examples.foundational.support.runner import configure from examples.foundational.support.runner import configure
async def main(room_url): async def main(room_url):
async with aiohttp.ClientSession() as session: async with aiohttp.ClientSession() as session:
# create a transport service object using environment variables for # create a transport service object using environment variables for
@@ -25,11 +27,23 @@ async def main(room_url):
meeting_duration_minutes, meeting_duration_minutes,
mic_enabled=True mic_enabled=True
) )
tts = ElevenLabsTTSService(aiohttp_session=session, api_key=os.getenv("ELEVENLABS_API_KEY"), voice_id=os.getenv("ELEVENLABS_VOICE_ID"))
"""
tts = ElevenLabsTTSService(
aiohttp_session=session,
api_key=os.getenv("ELEVENLABS_API_KEY"),
voice_id=os.getenv("ELEVENLABS_VOICE_ID"))
"""
tts = PlayHTAIService(
api_key=os.getenv("PLAY_HT_API_KEY"),
user_id=os.getenv("PLAY_HT_USER_ID"),
voice_url=os.getenv("PLAY_HT_VOICE_URL"),
)
# Register an event handler so we can play the audio when the participant joins. # Register an event handler so we can play the audio when the participant joins.
@transport.event_handler("on_participant_joined") @transport.event_handler("on_participant_joined")
async def on_participant_joined(transport, participant): async def on_participant_joined(transport, participant):
nonlocal tts
if participant["info"]["isLocal"]: if participant["info"]["isLocal"]:
return return
@@ -42,6 +56,7 @@ async def main(room_url):
await transport.stop_when_done() await transport.stop_when_done()
await transport.run() await transport.run()
del(tts)
if __name__ == "__main__": if __name__ == "__main__":

View File

@@ -11,6 +11,7 @@ from dailyai.services.deepgram_ai_services import DeepgramTTSService
from dailyai.services.open_ai_services import OpenAILLMService from dailyai.services.open_ai_services import OpenAILLMService
from examples.foundational.support.runner import configure from examples.foundational.support.runner import configure
async def main(room_url): async def main(room_url):
async with aiohttp.ClientSession() as session: async with aiohttp.ClientSession() as session:
meeting_duration_minutes = 1 meeting_duration_minutes = 1
@@ -22,12 +23,18 @@ async def main(room_url):
mic_enabled=True mic_enabled=True
) )
tts = ElevenLabsTTSService(aiohttp_session=session, api_key=os.getenv("ELEVENLABS_API_KEY"), voice_id=os.getenv("ELEVENLABS_VOICE_ID")) tts = ElevenLabsTTSService(
aiohttp_session=session,
api_key=os.getenv("ELEVENLABS_API_KEY"),
voice_id=os.getenv("ELEVENLABS_VOICE_ID"))
# tts = AzureTTSService(api_key=os.getenv("AZURE_SPEECH_API_KEY"), region=os.getenv("AZURE_SPEECH_REGION")) # tts = AzureTTSService(api_key=os.getenv("AZURE_SPEECH_API_KEY"), region=os.getenv("AZURE_SPEECH_REGION"))
# tts = DeepgramTTSService(aiohttp_session=session, api_key=os.getenv("DEEPGRAM_API_KEY"), voice=os.getenv("DEEPGRAM_VOICE")) # tts = DeepgramTTSService(aiohttp_session=session, api_key=os.getenv("DEEPGRAM_API_KEY"), voice=os.getenv("DEEPGRAM_VOICE"))
llm = AzureLLMService(api_key=os.getenv("AZURE_CHATGPT_API_KEY"), endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"), model=os.getenv("AZURE_CHATGPT_MODEL")) llm = AzureLLMService(
#llm = OpenAILLMService(api_key=os.getenv("OPENAI_CHATGPT_API_KEY")) api_key=os.getenv("AZURE_CHATGPT_API_KEY"),
endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"),
model=os.getenv("AZURE_CHATGPT_MODEL"))
# llm = OpenAILLMService(api_key=os.getenv("OPENAI_CHATGPT_API_KEY"))
messages = [{ messages = [{
"role": "system", "role": "system",
"content": "You are an LLM in a WebRTC session, and this is a 'hello world' demo. Say hello to the world." "content": "You are an LLM in a WebRTC session, and this is a 'hello world' demo. Say hello to the world."

View File

@@ -28,7 +28,11 @@ async def main(room_url):
camera_height=1024 camera_height=1024
) )
imagegen = FalImageGenService(image_size="1024x1024", aiohttp_session=session, key_id=os.getenv("FAL_KEY_ID"), key_secret=os.getenv("FAL_KEY_SECRET")) imagegen = FalImageGenService(
image_size="1024x1024",
aiohttp_session=session,
key_id=os.getenv("FAL_KEY_ID"),
key_secret=os.getenv("FAL_KEY_SECRET"))
# imagegen = OpenAIImageGenService(aiohttp_session=session, api_key=os.getenv("OPENAI_DALLE_API_KEY"), image_size="1024x1024") # imagegen = OpenAIImageGenService(aiohttp_session=session, api_key=os.getenv("OPENAI_DALLE_API_KEY"), image_size="1024x1024")
# imagegen = AzureImageGenServiceREST(image_size="1024x1024", aiohttp_session=session, api_key=os.getenv("AZURE_DALLE_API_KEY"), endpoint=os.getenv("AZURE_DALLE_ENDPOINT"), model=os.getenv("AZURE_DALLE_MODEL")) # imagegen = AzureImageGenServiceREST(image_size="1024x1024", aiohttp_session=session, api_key=os.getenv("AZURE_DALLE_API_KEY"), endpoint=os.getenv("AZURE_DALLE_ENDPOINT"), model=os.getenv("AZURE_DALLE_MODEL"))

View File

@@ -10,6 +10,7 @@ from dailyai.services.elevenlabs_ai_service import ElevenLabsTTSService
from examples.foundational.support.runner import configure from examples.foundational.support.runner import configure
async def main(room_url: str): async def main(room_url: str):
async with aiohttp.ClientSession() as session: async with aiohttp.ClientSession() as session:
transport = DailyTransportService( transport = DailyTransportService(
@@ -22,9 +23,17 @@ async def main(room_url: str):
camera_enabled=False camera_enabled=False
) )
llm = AzureLLMService(api_key=os.getenv("AZURE_CHATGPT_API_KEY"), endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"), model=os.getenv("AZURE_CHATGPT_MODEL")) llm = AzureLLMService(
azure_tts = AzureTTSService(api_key=os.getenv("AZURE_SPEECH_API_KEY"), region=os.getenv("AZURE_SPEECH_REGION")) api_key=os.getenv("AZURE_CHATGPT_API_KEY"),
elevenlabs_tts = ElevenLabsTTSService(aiohttp_session=session, api_key=os.getenv("ELEVENLABS_API_KEY"), voice_id=os.getenv("ELEVENLABS_VOICE_ID")) endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"),
model=os.getenv("AZURE_CHATGPT_MODEL"))
azure_tts = AzureTTSService(
api_key=os.getenv("AZURE_SPEECH_API_KEY"),
region=os.getenv("AZURE_SPEECH_REGION"))
elevenlabs_tts = ElevenLabsTTSService(
aiohttp_session=session,
api_key=os.getenv("ELEVENLABS_API_KEY"),
voice_id=os.getenv("ELEVENLABS_VOICE_ID"))
messages = [{"role": "system", "content": "tell the user a joke about llamas"}] messages = [{"role": "system", "content": "tell the user a joke about llamas"}]

View File

@@ -11,6 +11,7 @@ from dailyai.services.open_ai_services import OpenAIImageGenService
from examples.foundational.support.runner import configure from examples.foundational.support.runner import configure
async def main(room_url): async def main(room_url):
async with aiohttp.ClientSession() as session: async with aiohttp.ClientSession() as session:
meeting_duration_minutes = 5 meeting_duration_minutes = 5
@@ -26,11 +27,21 @@ async def main(room_url):
camera_height=1024 camera_height=1024
) )
llm = AzureLLMService(api_key=os.getenv("AZURE_CHATGPT_API_KEY"), endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"), model=os.getenv("AZURE_CHATGPT_MODEL")) llm = AzureLLMService(
tts = ElevenLabsTTSService(aiohttp_session=session, api_key=os.getenv("ELEVENLABS_API_KEY"), voice_id="ErXwobaYiN019PkySvjV") api_key=os.getenv("AZURE_CHATGPT_API_KEY"),
endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"),
model=os.getenv("AZURE_CHATGPT_MODEL"))
tts = ElevenLabsTTSService(
aiohttp_session=session,
api_key=os.getenv("ELEVENLABS_API_KEY"),
voice_id="ErXwobaYiN019PkySvjV")
# tts = AzureTTSService(api_key=os.getenv("AZURE_SPEECH_API_KEY"), region=os.getenv("AZURE_SPEECH_REGION")) # tts = AzureTTSService(api_key=os.getenv("AZURE_SPEECH_API_KEY"), region=os.getenv("AZURE_SPEECH_REGION"))
dalle = FalImageGenService(image_size="1024x1024", aiohttp_session=session, key_id=os.getenv("FAL_KEY_ID"), key_secret=os.getenv("FAL_KEY_SECRET")) dalle = FalImageGenService(
image_size="1024x1024",
aiohttp_session=session,
key_id=os.getenv("FAL_KEY_ID"),
key_secret=os.getenv("FAL_KEY_SECRET"))
# dalle = OpenAIImageGenService(aiohttp_session=session, api_key=os.getenv("OPENAI_DALLE_API_KEY"), image_size="1024x1024") # dalle = OpenAIImageGenService(aiohttp_session=session, api_key=os.getenv("OPENAI_DALLE_API_KEY"), image_size="1024x1024")
# dalle = AzureImageGenServiceREST(image_size="1024x1024", aiohttp_session=session, api_key=os.getenv("AZURE_DALLE_API_KEY"), endpoint=os.getenv("AZURE_DALLE_ENDPOINT"), model=os.getenv("AZURE_DALLE_MODEL")) # dalle = AzureImageGenServiceREST(image_size="1024x1024", aiohttp_session=session, api_key=os.getenv("AZURE_DALLE_API_KEY"), endpoint=os.getenv("AZURE_DALLE_ENDPOINT"), model=os.getenv("AZURE_DALLE_MODEL"))

View File

@@ -88,11 +88,12 @@ Once you have collected all of that information, respond with a JSON object cont
) )
) )
) )
transport.transcription_settings["extra"]["endpointing"] = True
transport.transcription_settings["extra"]["punctuate"] = True transport.transcription_settings["extra"]["punctuate"] = True
await asyncio.gather(transport.run(), handle_transcriptions()) await asyncio.gather(transport.run(), handle_transcriptions())
if __name__ == "__main__": if __name__ == "__main__":
(url, token) = configure() (url, token) = configure()
asyncio.run(main(url, token)) asyncio.run(main(url, token))

View File

@@ -18,6 +18,7 @@ from dailyai.services.fal_ai_services import FalImageGenService
from examples.foundational.support.runner import configure from examples.foundational.support.runner import configure
class ImageSyncAggregator(AIService): class ImageSyncAggregator(AIService):
def __init__(self, speaking_path: str, waiting_path: str): def __init__(self, speaking_path: str, waiting_path: str):
self._speaking_image = Image.open(speaking_path) self._speaking_image = Image.open(speaking_path)
@@ -46,9 +47,18 @@ async def main(room_url: str, token):
transport._mic_enabled = True transport._mic_enabled = True
transport._mic_sample_rate = 16000 transport._mic_sample_rate = 16000
llm = AzureLLMService(api_key=os.getenv("AZURE_CHATGPT_API_KEY"), endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"), model=os.getenv("AZURE_CHATGPT_MODEL")) llm = AzureLLMService(
tts = AzureTTSService(api_key=os.getenv("AZURE_SPEECH_API_KEY"), region=os.getenv("AZURE_SPEECH_REGION")) api_key=os.getenv("AZURE_CHATGPT_API_KEY"),
img = FalImageGenService(image_size="1024x1024", aiohttp_session=session, key_id=os.getenv("FAL_KEY_ID"), key_secret=os.getenv("FAL_KEY_SECRET")) endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"),
model=os.getenv("AZURE_CHATGPT_MODEL"))
tts = AzureTTSService(
api_key=os.getenv("AZURE_SPEECH_API_KEY"),
region=os.getenv("AZURE_SPEECH_REGION"))
img = FalImageGenService(
image_size="1024x1024",
aiohttp_session=session,
key_id=os.getenv("FAL_KEY_ID"),
key_secret=os.getenv("FAL_KEY_SECRET"))
async def get_images(): async def get_images():
get_speaking_task = asyncio.create_task( get_speaking_task = asyncio.create_task(

View File

@@ -25,10 +25,13 @@ async def main(room_url: str, token):
camera_enabled=False, camera_enabled=False,
) )
llm = OpenAILLMService(api_key=os.getenv( llm = AzureLLMService(
"OPENAI_CHATGPT_API_KEY"), model="gpt-4", tools=None) api_key=os.getenv("AZURE_CHATGPT_API_KEY"),
tts = AzureTTSService(api_key=os.getenv( endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"),
"AZURE_SPEECH_API_KEY"), region=os.getenv("AZURE_SPEECH_REGION")) model=os.getenv("AZURE_CHATGPT_MODEL"))
tts = AzureTTSService(
api_key=os.getenv("AZURE_SPEECH_API_KEY"),
region=os.getenv("AZURE_SPEECH_REGION"))
async def run_response(user_speech, tma_in, tma_out): async def run_response(user_speech, tma_in, tma_out):
await tts.run_to_queue( await tts.run_to_queue(

View File

@@ -10,7 +10,8 @@ from dailyai.queue_frame import AudioQueueFrame, ImageQueueFrame
from examples.foundational.support.runner import configure from examples.foundational.support.runner import configure
async def main(room_url:str):
async def main(room_url: str):
async with aiohttp.ClientSession() as session: async with aiohttp.ClientSession() as session:
transport = DailyTransportService( transport = DailyTransportService(
room_url, room_url,
@@ -24,16 +25,30 @@ async def main(room_url:str):
camera_height=1024 camera_height=1024
) )
llm = AzureLLMService(api_key=os.getenv("AZURE_CHATGPT_API_KEY"), endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"), model=os.getenv("AZURE_CHATGPT_MODEL")) llm = AzureLLMService(
tts1 = AzureTTSService(api_key=os.getenv("AZURE_SPEECH_API_KEY"), region=os.getenv("AZURE_SPEECH_REGION")) api_key=os.getenv("AZURE_CHATGPT_API_KEY"),
tts2 = ElevenLabsTTSService(aiohttp_session=session, api_key=os.getenv("ELEVENLABS_API_KEY"), voice_id="jBpfuIE2acCO8z3wKNLl") endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"),
dalle = FalImageGenService(image_size="1024x1024", aiohttp_session=session, key_id=os.getenv("FAL_KEY_ID"), key_secret=os.getenv("FAL_KEY_SECRET")) model=os.getenv("AZURE_CHATGPT_MODEL"))
tts1 = AzureTTSService(
api_key=os.getenv("AZURE_SPEECH_API_KEY"),
region=os.getenv("AZURE_SPEECH_REGION"))
tts2 = ElevenLabsTTSService(
aiohttp_session=session,
api_key=os.getenv("ELEVENLABS_API_KEY"),
voice_id="jBpfuIE2acCO8z3wKNLl")
dalle = FalImageGenService(
image_size="1024x1024",
aiohttp_session=session,
key_id=os.getenv("FAL_KEY_ID"),
key_secret=os.getenv("FAL_KEY_SECRET"))
bot1_messages = [ bot1_messages = [
{"role": "system", "content": "You are a stern librarian. You strongly believe that a hot dog is a sandwich. Start by stating this fact in a few sentences, then be prepared to debate this with the user. You shouldn't ever compromise on the fundamental truth that a hot dog is a sandwich. Your responses should only be a few sentences long."}, {"role": "system", "content": "You are a stern librarian. You strongly believe that a hot dog is a sandwich. Start by stating this fact in a few sentences, then be prepared to debate this with the user. You shouldn't ever compromise on the fundamental truth that a hot dog is a sandwich. Your responses should only be a few sentences long."},
] ]
bot2_messages = [ bot2_messages = [
{"role": "system", "content": "You are a silly cat, and you strongly believe that a hot dog is not a sandwich. Debate this with the user, only responding with a few sentences. Don't ever accept that a hot dog is a sandwich."}, {
"role": "system",
"content": "You are a silly cat, and you strongly believe that a hot dog is not a sandwich. Debate this with the user, only responding with a few sentences. Don't ever accept that a hot dog is a sandwich."},
] ]
async def get_bot1_statement(): async def get_bot1_statement():

View File

@@ -71,7 +71,7 @@ class TranscriptFilter(AIService):
class NameCheckFilter(AIService): class NameCheckFilter(AIService):
def __init__(self, names:list[str]): def __init__(self, names: list[str]):
self.names = names self.names = names
self.sentence = "" self.sentence = ""
@@ -123,8 +123,14 @@ async def main(room_url: str, token):
transport._camera_width = 720 transport._camera_width = 720
transport._camera_height = 1280 transport._camera_height = 1280
llm = AzureLLMService(api_key=os.getenv("AZURE_CHATGPT_API_KEY"), endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"), model=os.getenv("AZURE_CHATGPT_MODEL")) llm = AzureLLMService(
tts = ElevenLabsTTSService(aiohttp_session=session, api_key=os.getenv("ELEVENLABS_API_KEY"), voice_id="jBpfuIE2acCO8z3wKNLl") api_key=os.getenv("AZURE_CHATGPT_API_KEY"),
endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"),
model=os.getenv("AZURE_CHATGPT_MODEL"))
tts = ElevenLabsTTSService(
aiohttp_session=session,
api_key=os.getenv("ELEVENLABS_API_KEY"),
voice_id="jBpfuIE2acCO8z3wKNLl")
isa = ImageSyncAggregator() isa = ImageSyncAggregator()
@transport.event_handler("on_first_other_participant_joined") @transport.event_handler("on_first_other_participant_joined")

View File

@@ -14,7 +14,7 @@ from typing import AsyncGenerator
from examples.foundational.support.runner import configure from examples.foundational.support.runner import configure
logging.basicConfig(format=f"%(levelno)s %(asctime)s %(message)s") # or whatever logging.basicConfig(format=f"%(levelno)s %(asctime)s %(message)s") # or whatever
logger = logging.getLogger("dailyai") logger = logging.getLogger("dailyai")
logger.setLevel(logging.DEBUG) logger.setLevel(logging.DEBUG)
@@ -36,8 +36,6 @@ for file in sound_files:
sounds[file] = audio_file.readframes(-1) sounds[file] = audio_file.readframes(-1)
class OutboundSoundEffectWrapper(AIService): class OutboundSoundEffectWrapper(AIService):
def __init__(self): def __init__(self):
pass pass
@@ -50,6 +48,7 @@ class OutboundSoundEffectWrapper(AIService):
else: else:
yield frame yield frame
class InboundSoundEffectWrapper(AIService): class InboundSoundEffectWrapper(AIService):
def __init__(self): def __init__(self):
pass pass
@@ -75,14 +74,20 @@ async def main(room_url: str, token):
camera_enabled=False camera_enabled=False
) )
llm = AzureLLMService(api_key=os.getenv("AZURE_CHATGPT_API_KEY"), endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"), model=os.getenv("AZURE_CHATGPT_MODEL")) llm = AzureLLMService(
tts = ElevenLabsTTSService(aiohttp_session=session, api_key=os.getenv("ELEVENLABS_API_KEY"), voice_id="ErXwobaYiN019PkySvjV") api_key=os.getenv("AZURE_CHATGPT_API_KEY"),
endpoint=os.getenv("AZURE_CHATGPT_ENDPOINT"),
model=os.getenv("AZURE_CHATGPT_MODEL"))
tts = ElevenLabsTTSService(
aiohttp_session=session,
api_key=os.getenv("ELEVENLABS_API_KEY"),
voice_id="ErXwobaYiN019PkySvjV")
@transport.event_handler("on_first_other_participant_joined") @transport.event_handler("on_first_other_participant_joined")
async def on_first_other_participant_joined(transport): async def on_first_other_participant_joined(transport):
await tts.say("Hi, I'm listening!", transport.send_queue) await tts.say("Hi, I'm listening!", transport.send_queue)
await transport.send_queue.put(AudioQueueFrame(sounds["ding1.wav"])) await transport.send_queue.put(AudioQueueFrame(sounds["ding1.wav"]))
async def handle_transcriptions(): async def handle_transcriptions():
messages = [ messages = [
{"role": "system", "content": "You are a helpful LLM in a WebRTC call. Your goal is to demonstrate your capabilities in a succinct way. Your output will be converted to audio. Respond to what the user said in a creative and helpful way."}, {"role": "system", "content": "You are a helpful LLM in a WebRTC call. Your goal is to demonstrate your capabilities in a succinct way. Your output will be converted to audio. Respond to what the user said in a creative and helpful way."},
@@ -117,7 +122,6 @@ async def main(room_url: str, token):
) )
) )
transport.transcription_settings["extra"]["punctuate"] = True transport.transcription_settings["extra"]["punctuate"] = True
await asyncio.gather(transport.run(), handle_transcriptions()) await asyncio.gather(transport.run(), handle_transcriptions())

View File

@@ -5,6 +5,7 @@ from dailyai.services.whisper_ai_services import WhisperSTTService
from examples.foundational.support.runner import configure from examples.foundational.support.runner import configure
async def main(room_url: str): async def main(room_url: str):
transport = DailyTransportService( transport = DailyTransportService(
room_url, room_url,

View File

@@ -17,7 +17,7 @@ async def main(room_url: str):
camera_enabled=False, camera_enabled=False,
speaker_enabled=True, speaker_enabled=True,
duration_minutes=meeting_duration_minutes, duration_minutes=meeting_duration_minutes,
start_transcription = True start_transcription=True
) )
stt = WhisperSTTService() stt = WhisperSTTService()
transcription_output_queue = asyncio.Queue() transcription_output_queue = asyncio.Queue()

View File

@@ -7,6 +7,7 @@ import requests
from dotenv import load_dotenv from dotenv import load_dotenv
load_dotenv() load_dotenv()
def configure(): def configure():
parser = argparse.ArgumentParser(description="Daily AI SDK Bot Sample") parser = argparse.ArgumentParser(description="Daily AI SDK Bot Sample")
parser.add_argument( parser.add_argument(
@@ -26,12 +27,12 @@ def configure():
key = args.apikey or os.getenv("DAILY_API_KEY") key = args.apikey or os.getenv("DAILY_API_KEY")
if not url: if not url:
raise Exception("No Daily room specified. use the -u/--url option from the command line, or set DAILY_SAMPLE_ROOM_URL in your environment to specify a Daily room URL.") raise Exception(
"No Daily room specified. use the -u/--url option from the command line, or set DAILY_SAMPLE_ROOM_URL in your environment to specify a Daily room URL.")
if not key: if not key:
raise Exception("No Daily API key specified. use the -k/--apikey option from the command line, or set DAILY_API_KEY in your environment to specify a Daily API key, available from https://dashboard.daily.co/developers.") raise Exception("No Daily API key specified. use the -k/--apikey option from the command line, or set DAILY_API_KEY in your environment to specify a Daily API key, available from https://dashboard.daily.co/developers.")
# Create a meeting token for the given room with an expiration 1 hour in the future. # Create a meeting token for the given room with an expiration 1 hour in the future.
room_name: str = urllib.parse.urlparse(url).path[1:] room_name: str = urllib.parse.urlparse(url).path[1:]
expiration: float = time.time() + 60 * 60 expiration: float = time.time() + 60 * 60

View File

@@ -30,8 +30,6 @@ for file in sound_files:
sounds[file] = audio_file.readframes(-1) sounds[file] = audio_file.readframes(-1)
class OutboundSoundEffectWrapper(AIService): class OutboundSoundEffectWrapper(AIService):
def __init__(self): def __init__(self):
pass pass
@@ -44,6 +42,7 @@ class OutboundSoundEffectWrapper(AIService):
else: else:
yield frame yield frame
class InboundSoundEffectWrapper(AIService): class InboundSoundEffectWrapper(AIService):
def __init__(self): def __init__(self):
pass pass
@@ -81,6 +80,7 @@ async def main(room_url: str, token, phone):
async def on_first_other_participant_joined(transport): async def on_first_other_participant_joined(transport):
await tts.say("Hi, I'm listening!", transport.send_queue) await tts.say("Hi, I'm listening!", transport.send_queue)
await transport.send_queue.put(AudioQueueFrame(sounds["ding1.wav"])) await transport.send_queue.put(AudioQueueFrame(sounds["ding1.wav"]))
async def handle_transcriptions(): async def handle_transcriptions():
messages = [ messages = [
{"role": "system", "content": "You are a helpful LLM in a WebRTC call. Your goal is to demonstrate your capabilities in a succinct way. Your output will be converted to audio. Respond to what the user said in a creative and helpful way."}, {"role": "system", "content": "You are a helpful LLM in a WebRTC call. Your goal is to demonstrate your capabilities in a succinct way. Your output will be converted to audio. Respond to what the user said in a creative and helpful way."},
@@ -124,7 +124,6 @@ async def main(room_url: str, token, phone):
transport.start_recording() transport.start_recording()
transport.dialout(phone) transport.dialout(phone)
transport.transcription_settings["extra"]["punctuate"] = True transport.transcription_settings["extra"]["punctuate"] = True
await asyncio.gather(transport.run(), handle_transcriptions()) await asyncio.gather(transport.run(), handle_transcriptions())