refactor party tonight
This commit is contained in:
@@ -2,9 +2,11 @@ import argparse
|
||||
import asyncio
|
||||
import re
|
||||
|
||||
from dailyai.services.ai_services import SentenceAggregator
|
||||
from dailyai.services.daily_transport_service import DailyTransportService
|
||||
from dailyai.services.azure_ai_services import AzureLLMService, AzureTTSService
|
||||
from dailyai.queue_frame import QueueFrame, FrameType
|
||||
from dailyai.services.elevenlabs_ai_service import ElevenLabsTTSService
|
||||
|
||||
async def main(room_url:str):
|
||||
global transport
|
||||
@@ -22,34 +24,46 @@ async def main(room_url:str):
|
||||
transport.camera_enabled = False
|
||||
|
||||
llm = AzureLLMService()
|
||||
tts = AzureTTSService()
|
||||
azure_tts = AzureTTSService()
|
||||
elevenlabs_tts = ElevenLabsTTSService(voice_id="ErXwobaYiN019PkySvjV")
|
||||
|
||||
messages = [{"role": "system", "content": "tell the user a joke about llamas"}]
|
||||
|
||||
# Start a task to run the LLM to create a joke, and convert the LLM output to audio frames. This task
|
||||
# will run in parallel with generating and speaking the audio for static text, so there's no delay to
|
||||
# speak the LLM response.
|
||||
buffer_queue = asyncio.Queue()
|
||||
llm_response_task = asyncio.create_task(
|
||||
elevenlabs_tts.run_to_queue(
|
||||
buffer_queue,
|
||||
SentenceAggregator().run(
|
||||
llm.run([QueueFrame(FrameType.LLM_MESSAGE, messages)])
|
||||
),
|
||||
True,
|
||||
)
|
||||
)
|
||||
|
||||
@transport.event_handler("on_participant_joined")
|
||||
async def on_joined(transport, participant):
|
||||
if participant["id"] == transport.my_participant_id:
|
||||
return
|
||||
|
||||
# queue two pieces of speech: one specified as a text literal,
|
||||
# and one generated by an llm. We'll kick off the llm first, and let
|
||||
# it generate a response while we're speaking the literal string.
|
||||
#
|
||||
# Note that in this case, we don't use `run_llm_async` because we're
|
||||
# taking advantage of the time spent speaking the first phrase to generate
|
||||
# the entire LLM response, and this happens asynchronously in a task.
|
||||
llm_response_task = asyncio.create_task(llm.run_llm(
|
||||
[{"role": "system", "content": "tell the user a joke about llamas"}]
|
||||
))
|
||||
await azure_tts.run_to_queue(
|
||||
transport.send_queue,
|
||||
[QueueFrame(FrameType.SENTENCE, "My friend the LLM is now going to tell a joke about llamas.")]
|
||||
)
|
||||
|
||||
async for audio_chunk in tts.run_tts("My friend the LLM is now going to tell a joke about llamas."):
|
||||
transport.output_queue.put(QueueFrame(FrameType.AUDIO, audio_chunk))
|
||||
async def buffer_to_send_queue():
|
||||
while True:
|
||||
frame = await buffer_queue.get()
|
||||
await transport.send_queue.put(frame)
|
||||
buffer_queue.task_done()
|
||||
if frame.frame_type == FrameType.END_STREAM:
|
||||
break
|
||||
|
||||
llm_response = await llm_response_task
|
||||
async for audio_chunk in tts.run_tts(llm_response):
|
||||
transport.output_queue.put(QueueFrame(FrameType.AUDIO, audio_chunk))
|
||||
await asyncio.gather(llm_response_task, buffer_to_send_queue())
|
||||
|
||||
|
||||
# wait for the output queue to be empty, then leave the meeting
|
||||
transport.output_queue.join()
|
||||
transport.wait_for_send_queue_to_empty()
|
||||
transport.stop()
|
||||
|
||||
await transport.run()
|
||||
@@ -61,6 +75,6 @@ if __name__ == "__main__":
|
||||
"-u", "--url", type=str, required=True, help="URL of the Daily room to join"
|
||||
)
|
||||
|
||||
args: argparse.Namespace = parser.parse_args()
|
||||
args, unknown = parser.parse_known_args()
|
||||
|
||||
asyncio.run(main(args.url))
|
||||
|
||||
Reference in New Issue
Block a user