Update 05- sample, sticking with generators here because they give us a lot more control over the order that things get queued.
This commit is contained in:
@@ -9,7 +9,7 @@ from dailyai.services.ai_services import TTSService
|
|||||||
|
|
||||||
|
|
||||||
class ElevenLabsTTSService(TTSService):
|
class ElevenLabsTTSService(TTSService):
|
||||||
def __init__(self, input_queue, output_queue, api_key=None, voice_id=None):
|
def __init__(self, input_queue=None, output_queue=None, api_key=None, voice_id=None):
|
||||||
super().__init__(input_queue, output_queue)
|
super().__init__(input_queue, output_queue)
|
||||||
|
|
||||||
self.api_key = api_key or os.getenv("ELEVENLABS_API_KEY")
|
self.api_key = api_key or os.getenv("ELEVENLABS_API_KEY")
|
||||||
|
|||||||
@@ -66,9 +66,14 @@ class OpenAIImageGenService(ImageGenService):
|
|||||||
size=size
|
size=size
|
||||||
)
|
)
|
||||||
image_url = image.data[0].url
|
image_url = image.data[0].url
|
||||||
response = requests.get(image_url)
|
if not image_url:
|
||||||
|
raise Exception("No image provided in response", image)
|
||||||
|
|
||||||
dalle_stream = io.BytesIO(response.content)
|
# Load the image from the url
|
||||||
dalle_im = Image.open(dalle_stream)
|
async with aiohttp.ClientSession() as session:
|
||||||
|
async with session.get(image_url) as response:
|
||||||
|
image_stream = io.BytesIO(await response.content.read())
|
||||||
|
image = Image.open(image_stream)
|
||||||
|
return (image_url, image.tobytes())
|
||||||
|
|
||||||
return (image_url, dalle_im.tobytes())
|
return (image_url, dalle_im.tobytes())
|
||||||
|
|||||||
@@ -32,6 +32,10 @@ async def main(room_url:str):
|
|||||||
# queue two pieces of speech: one specified as a text literal,
|
# queue two pieces of speech: one specified as a text literal,
|
||||||
# and one generated by an llm. We'll kick off the llm first, and let
|
# and one generated by an llm. We'll kick off the llm first, and let
|
||||||
# it generate a response while we're speaking the literal string.
|
# it generate a response while we're speaking the literal string.
|
||||||
|
#
|
||||||
|
# Note that in this case, we don't use `run_llm_async` because we're
|
||||||
|
# taking advantage of the time spent speaking the first phrase to generate
|
||||||
|
# the entire LLM response, and this happens asynchronously in a task.
|
||||||
llm_response_task = asyncio.create_task(llm.run_llm(
|
llm_response_task = asyncio.create_task(llm.run_llm(
|
||||||
[{"role": "system", "content": "tell the user a joke about llamas"}]
|
[{"role": "system", "content": "tell the user a joke about llamas"}]
|
||||||
))
|
))
|
||||||
|
|||||||
@@ -25,7 +25,7 @@ async def main(room_url):
|
|||||||
transport.camera_height = 1024
|
transport.camera_height = 1024
|
||||||
|
|
||||||
llm = AzureLLMService()
|
llm = AzureLLMService()
|
||||||
tts = ElevenLabsTTSService()
|
tts = ElevenLabsTTSService(voice_id="ErXwobaYiN019PkySvjV")
|
||||||
dalle = OpenAIImageGenService()
|
dalle = OpenAIImageGenService()
|
||||||
|
|
||||||
# Get a complete audio chunk from the given text. Splitting this into its own
|
# Get a complete audio chunk from the given text. Splitting this into its own
|
||||||
@@ -39,9 +39,9 @@ async def main(room_url):
|
|||||||
|
|
||||||
async def get_month_data(month):
|
async def get_month_data(month):
|
||||||
image_text = ""
|
image_text = ""
|
||||||
current_clause = ""
|
|
||||||
tts_tasks = []
|
tts_tasks = []
|
||||||
async for text in llm.run_llm_async(
|
first_sentence = True
|
||||||
|
async for sentence in llm.run_llm_async_sentences(
|
||||||
[
|
[
|
||||||
{
|
{
|
||||||
"role": "system",
|
"role": "system",
|
||||||
@@ -49,18 +49,24 @@ async def main(room_url):
|
|||||||
}
|
}
|
||||||
]
|
]
|
||||||
):
|
):
|
||||||
image_text += text
|
image_text += sentence
|
||||||
current_clause += text
|
|
||||||
if re.match(r"^.*[.!?]$", text):
|
if first_sentence:
|
||||||
tts_tasks.append(get_all_audio(current_clause))
|
sentence = f"{month}: {sentence}"
|
||||||
current_clause = ""
|
else:
|
||||||
|
first_sentence = False
|
||||||
|
|
||||||
|
tts_tasks.append(get_all_audio(sentence))
|
||||||
|
|
||||||
tts_tasks.insert(0, dalle.run_image_gen(image_text, "1024x1024"))
|
tts_tasks.insert(0, dalle.run_image_gen(image_text, "1024x1024"))
|
||||||
|
|
||||||
|
print(f"waiting for tasks to finish for {month}")
|
||||||
data = await asyncio.gather(
|
data = await asyncio.gather(
|
||||||
*tts_tasks
|
*tts_tasks
|
||||||
)
|
)
|
||||||
|
|
||||||
|
print(f"done gathering tts tasks for {month}")
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"month": month,
|
"month": month,
|
||||||
"text": image_text,
|
"text": image_text,
|
||||||
@@ -83,11 +89,8 @@ async def main(room_url):
|
|||||||
"December",
|
"December",
|
||||||
]
|
]
|
||||||
|
|
||||||
@transport.event_handler("on_participant_joined")
|
@transport.event_handler("on_first_other_participant_joined")
|
||||||
async def on_participant_joined(transport, participant):
|
async def on_first_other_participant_joined(transport):
|
||||||
if participant["id"] == transport.my_participant_id:
|
|
||||||
return
|
|
||||||
|
|
||||||
# This will play the months in the order they're completed. The benefit
|
# This will play the months in the order they're completed. The benefit
|
||||||
# is we'll have as little delay as possible before the first month, and
|
# is we'll have as little delay as possible before the first month, and
|
||||||
# likely no delay between months, but the months won't display in order.
|
# likely no delay between months, but the months won't display in order.
|
||||||
@@ -116,6 +119,6 @@ if __name__=="__main__":
|
|||||||
"-u", "--url", type=str, required=True, help="URL of the Daily room to join"
|
"-u", "--url", type=str, required=True, help="URL of the Daily room to join"
|
||||||
)
|
)
|
||||||
|
|
||||||
args: argparse.Namespace = parser.parse_args()
|
args, unknown = parser.parse_known_args()
|
||||||
|
|
||||||
asyncio.run(main(args.url))
|
asyncio.run(main(args.url))
|
||||||
|
|||||||
Reference in New Issue
Block a user