scripts(evals): simplify eval configuration and allow RunnerArgs body

This commit is contained in:
Aleix Conchillo Flaqué
2025-10-29 15:17:36 -07:00
parent 3b3a215155
commit a997655eac
2 changed files with 158 additions and 160 deletions

View File

@@ -10,9 +10,10 @@ import os
import re import re
import time import time
import wave import wave
from dataclasses import dataclass
from datetime import datetime from datetime import datetime
from pathlib import Path from pathlib import Path
from typing import List, Optional, Tuple from typing import Any, List, Optional, Tuple
import aiofiles import aiofiles
from deepgram import LiveOptions from deepgram import LiveOptions
@@ -53,6 +54,14 @@ EVAL_TIMEOUT_SECS = 120
EvalPrompt = str | Tuple[str, ImageFile] EvalPrompt = str | Tuple[str, ImageFile]
@dataclass
class EvalConfig:
prompt: EvalPrompt
eval: str
eval_speaks_first: bool = False
runner_args_body: Optional[Any] = None
class EvalRunner: class EvalRunner:
def __init__( def __init__(
self, self,
@@ -93,9 +102,7 @@ class EvalRunner:
async def run_eval( async def run_eval(
self, self,
example_file: str, example_file: str,
prompt: EvalPrompt, eval_config: EvalConfig,
eval: str,
user_speaks_first: bool = False,
): ):
if not re.match(self._pattern, example_file): if not re.match(self._pattern, example_file):
return return
@@ -112,10 +119,8 @@ class EvalRunner:
try: try:
tasks = [ tasks = [
asyncio.create_task(run_example_pipeline(script_path)), asyncio.create_task(run_example_pipeline(script_path, eval_config)),
asyncio.create_task( asyncio.create_task(run_eval_pipeline(self, example_file, eval_config)),
run_eval_pipeline(self, example_file, prompt, eval, user_speaks_first)
),
] ]
_, pending = await asyncio.wait(tasks, timeout=EVAL_TIMEOUT_SECS) _, pending = await asyncio.wait(tasks, timeout=EVAL_TIMEOUT_SECS)
if pending: if pending:
@@ -177,7 +182,7 @@ class EvalRunner:
return os.path.join(self._recordings_dir, f"{base_name}.wav") return os.path.join(self._recordings_dir, f"{base_name}.wav")
async def run_example_pipeline(script_path: Path): async def run_example_pipeline(script_path: Path, eval_config: EvalConfig):
room_url = os.getenv("DAILY_SAMPLE_ROOM_URL") room_url = os.getenv("DAILY_SAMPLE_ROOM_URL")
module = load_module_from_path(script_path) module = load_module_from_path(script_path)
@@ -196,6 +201,7 @@ async def run_example_pipeline(script_path: Path):
runner_args = RunnerArguments() runner_args = RunnerArguments()
runner_args.pipeline_idle_timeout_secs = PIPELINE_IDLE_TIMEOUT_SECS runner_args.pipeline_idle_timeout_secs = PIPELINE_IDLE_TIMEOUT_SECS
runner_args.body = eval_config.runner_args_body
await module.run_bot(transport, runner_args) await module.run_bot(transport, runner_args)
@@ -203,9 +209,7 @@ async def run_example_pipeline(script_path: Path):
async def run_eval_pipeline( async def run_eval_pipeline(
eval_runner: EvalRunner, eval_runner: EvalRunner,
example_file: str, example_file: str,
prompt: EvalPrompt, eval_config: EvalConfig,
eval: str,
user_speaks_first: bool = False,
): ):
logger.info(f"Starting eval bot") logger.info(f"Starting eval bot")
@@ -262,17 +266,17 @@ async def run_eval_pipeline(
# Load example prompt depending on image. # Load example prompt depending on image.
example_prompt = "" example_prompt = ""
example_image: Optional[ImageFile] = None example_image: Optional[ImageFile] = None
if isinstance(prompt, str): if isinstance(eval_config.prompt, str):
example_prompt = prompt example_prompt = eval_config.prompt
elif isinstance(prompt, tuple): elif isinstance(eval_config.prompt, tuple):
example_prompt, example_image = prompt example_prompt, example_image = eval_config.prompt
eval_prompt = f"The answer is correct if it matches: {eval}." eval_prompt = f"The answer is correct if it matches: {eval}."
common_system_prompt = ( common_system_prompt = (
"The user might say things other than the answer and that's allowed. " "The user might say things other than the answer and that's allowed. "
f"You should only call the eval function with your assessment when the user actually answers the question. {eval_prompt}" f"You should only call the eval function with your assessment when the user actually answers the question. {eval_prompt}"
) )
if user_speaks_first: if eval_config.eval_speaks_first:
system_prompt = f"You are an LLM eval, be extremly brief. You will start the conversation by saying: '{example_prompt}'. {common_system_prompt}" system_prompt = f"You are an LLM eval, be extremly brief. You will start the conversation by saying: '{example_prompt}'. {common_system_prompt}"
else: else:
system_prompt = f"You are an LLM eval, be extremly brief. Your goal is to first ask one question: {example_prompt}. {common_system_prompt}" system_prompt = f"You are an LLM eval, be extremly brief. Your goal is to first ask one question: {example_prompt}. {common_system_prompt}"
@@ -330,9 +334,9 @@ async def run_eval_pipeline(
# Default behavior is for the bot to speak first # Default behavior is for the bot to speak first
# If the eval bot speaks first, we append the prompt to the messages # If the eval bot speaks first, we append the prompt to the messages
if user_speaks_first: if eval_config.eval_speaks_first:
messages.append( messages.append(
{"role": "user", "content": f"Start by saying this exactly: '{prompt}'"} {"role": "user", "content": f"Start by saying this exactly: '{eval_config.prompt}'"}
) )
await task.queue_frames([LLMRunFrame()]) await task.queue_frames([LLMRunFrame()])

View File

@@ -11,7 +11,7 @@ from datetime import datetime, timezone
from pathlib import Path from pathlib import Path
from dotenv import load_dotenv from dotenv import load_dotenv
from eval import EvalRunner from eval import EvalConfig, EvalRunner
from loguru import logger from loguru import logger
from PIL import Image from PIL import Image
from utils import check_env_variables from utils import check_env_variables
@@ -24,190 +24,184 @@ ASSETS_DIR = SCRIPT_DIR / "assets"
FOUNDATIONAL_DIR = SCRIPT_DIR.parent.parent / "examples" / "foundational" FOUNDATIONAL_DIR = SCRIPT_DIR.parent.parent / "examples" / "foundational"
# Speaking order constants EVAL_SIMPLE_MATH = EvalConfig(
USER_SPEAKS_FIRST = True prompt="A simple math addition.",
BOT_SPEAKS_FIRST = False eval="Correct math addition.",
# Math
PROMPT_SIMPLE_MATH = "A simple math addition."
EVAL_SIMPLE_MATH = "Correct math addition."
# Weather
PROMPT_WEATHER = "What's the weather in San Francisco?"
EVAL_WEATHER = (
"Something specific about the current weather in San Francisco, including the degrees."
) )
# Online search EVAL_WEATHER = EvalConfig(
PROMPT_ONLINE_SEARCH = "What's the date right now in London?" prompt="What's the weather in San Francisco?",
EVAL_ONLINE_SEARCH = f"Today is {datetime.now(timezone.utc).strftime('%B %d, %Y')}." eval="Something specific about the current weather in San Francisco, including the degrees.",
)
# Switch language EVAL_ONLINE_SEARCH = EvalConfig(
PROMPT_SWITCH_LANGUAGE = "Say something in Spanish." prompt="What's the date right now in London?",
EVAL_SWITCH_LANGUAGE = "The user is now talking in Spanish." eval=f"Today is {datetime.now(timezone.utc).strftime('%B %d, %Y')}.",
)
# Vision EVAL_SWITCH_LANGUAGE = EvalConfig(
PROMPT_VISION = ("Briefly describe what you see.", Image.open(ASSETS_DIR / "cat.jpg")) prompt="Say something in Spanish.",
EVAL_VISION = "A cat description." eval="The user is now talking in Spanish.",
)
EVAL_VISION_CAMERA = EvalConfig(
prompt=("Briefly describe what you see.", Image.open(ASSETS_DIR / "cat.jpg")),
eval="A cat description.",
)
EVAL_VISION_IMAGE = EvalConfig(
prompt="Briefly describe this image.",
eval="A cat description.",
eval_speaks_first=True,
runner_args_body={
"image_path": ASSETS_DIR / "cat.jpg",
"question": "Briefly describe this image.",
},
)
EVAL_VOICEMAIL = EvalConfig(
prompt="Please leave a message after the beep.",
eval="Assess the conversation and determine if it is a voicemail.",
eval_speaks_first=True,
)
EVAL_CONVERSATION = EvalConfig(
prompt="Hello, this is Mark.",
eval="A start of a conversation, not a voicemail.",
eval_speaks_first=True,
)
# Voicemail
PROMPT_VOICEMAIL = "Please leave a message after the beep."
EVAL_VOICEMAIL = "Assess the conversation and determine if it is a voicemail."
PROMPT_CONVERSATION = "Hello, this is Mark."
EVAL_CONVERSATION = "A start of a conversation, not a voicemail."
TESTS_07 = [ TESTS_07 = [
# 07 series # 07 series
("07-interruptible.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07-interruptible.py", EVAL_SIMPLE_MATH),
("07-interruptible-cartesia-http.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07-interruptible-cartesia-http.py", EVAL_SIMPLE_MATH),
("07a-interruptible-speechmatics.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07a-interruptible-speechmatics.py", EVAL_SIMPLE_MATH),
("07aa-interruptible-soniox.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07aa-interruptible-soniox.py", EVAL_SIMPLE_MATH),
("07ab-interruptible-inworld-http.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07ab-interruptible-inworld-http.py", EVAL_SIMPLE_MATH),
("07ac-interruptible-asyncai.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07ac-interruptible-asyncai.py", EVAL_SIMPLE_MATH),
("07ac-interruptible-asyncai-http.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07ac-interruptible-asyncai-http.py", EVAL_SIMPLE_MATH),
("07b-interruptible-langchain.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07b-interruptible-langchain.py", EVAL_SIMPLE_MATH),
("07c-interruptible-deepgram.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07c-interruptible-deepgram.py", EVAL_SIMPLE_MATH),
("07c-interruptible-deepgram-flux.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07c-interruptible-deepgram-flux.py", EVAL_SIMPLE_MATH),
("07d-interruptible-elevenlabs.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07d-interruptible-elevenlabs.py", EVAL_SIMPLE_MATH),
( ("07d-interruptible-elevenlabs-http.py", EVAL_SIMPLE_MATH),
"07d-interruptible-elevenlabs-http.py", ("07f-interruptible-azure.py", EVAL_SIMPLE_MATH),
PROMPT_SIMPLE_MATH, ("07g-interruptible-openai.py", EVAL_SIMPLE_MATH),
EVAL_SIMPLE_MATH, ("07h-interruptible-openpipe.py", EVAL_SIMPLE_MATH),
BOT_SPEAKS_FIRST, ("07j-interruptible-gladia.py", EVAL_SIMPLE_MATH),
), ("07k-interruptible-lmnt.py", EVAL_SIMPLE_MATH),
("07f-interruptible-azure.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07l-interruptible-groq.py", EVAL_SIMPLE_MATH),
("07g-interruptible-openai.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07m-interruptible-aws.py", EVAL_SIMPLE_MATH),
("07h-interruptible-openpipe.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07m-interruptible-aws-strands.py", EVAL_WEATHER),
("07j-interruptible-gladia.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07n-interruptible-gemini.py", EVAL_SIMPLE_MATH),
("07k-interruptible-lmnt.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07n-interruptible-google.py", EVAL_SIMPLE_MATH),
("07l-interruptible-groq.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07o-interruptible-assemblyai.py", EVAL_SIMPLE_MATH),
("07m-interruptible-aws.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07q-interruptible-rime.py", EVAL_SIMPLE_MATH),
("07m-interruptible-aws-strands.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("07q-interruptible-rime-http.py", EVAL_SIMPLE_MATH),
("07n-interruptible-gemini.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07r-interruptible-riva-nim.py", EVAL_SIMPLE_MATH),
("07n-interruptible-google.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07s-interruptible-google-audio-in.py", EVAL_SIMPLE_MATH),
("07o-interruptible-assemblyai.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07t-interruptible-fish.py", EVAL_SIMPLE_MATH),
("07q-interruptible-rime.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07v-interruptible-neuphonic.py", EVAL_SIMPLE_MATH),
("07q-interruptible-rime-http.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07v-interruptible-neuphonic-http.py", EVAL_SIMPLE_MATH),
("07r-interruptible-riva-nim.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("07w-interruptible-fal.py", EVAL_SIMPLE_MATH),
( ("07y-interruptible-minimax.py", EVAL_SIMPLE_MATH),
"07s-interruptible-google-audio-in.py", ("07z-interruptible-sarvam.py", EVAL_SIMPLE_MATH),
PROMPT_SIMPLE_MATH, ("07ae-interruptible-hume.py", EVAL_SIMPLE_MATH),
EVAL_SIMPLE_MATH,
BOT_SPEAKS_FIRST,
),
("07t-interruptible-fish.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
("07v-interruptible-neuphonic.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
("07v-interruptible-neuphonic-http.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
("07w-interruptible-fal.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
("07y-interruptible-minimax.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
("07z-interruptible-sarvam.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
("07ae-interruptible-hume.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
# Needs a local XTTS docker instance running. # Needs a local XTTS docker instance running.
# ("07i-interruptible-xtts.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), # ("07i-interruptible-xtts.py", EVAL_SIMPLE_MATH),
# Needs a Krisp license. # Needs a Krisp license.
# ("07p-interruptible-krisp.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), # ("07p-interruptible-krisp.py", EVAL_SIMPLE_MATH),
# Needs GPU resources. # Needs GPU resources.
# ("07u-interruptible-ultravox.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), # ("07u-interruptible-ultravox.py", EVAL_SIMPLE_MATH),
]
TESTS_12 = [
("12-describe-image-openai.py", EVAL_VISION_IMAGE),
("12a-describe-image-anthropic.py", EVAL_VISION_IMAGE),
("12b-describe-image-aws.py", EVAL_VISION_IMAGE),
("12c-describe-image-gemini-flash.py", EVAL_VISION_IMAGE),
] ]
TESTS_14 = [ TESTS_14 = [
("14-function-calling.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14-function-calling.py", EVAL_WEATHER),
("14a-function-calling-anthropic.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14a-function-calling-anthropic.py", EVAL_WEATHER),
("14e-function-calling-google.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14e-function-calling-google.py", EVAL_WEATHER),
("14f-function-calling-groq.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14f-function-calling-groq.py", EVAL_WEATHER),
("14g-function-calling-grok.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14g-function-calling-grok.py", EVAL_WEATHER),
("14h-function-calling-azure.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14h-function-calling-azure.py", EVAL_WEATHER),
("14i-function-calling-fireworks.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14i-function-calling-fireworks.py", EVAL_WEATHER),
("14j-function-calling-nim.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14j-function-calling-nim.py", EVAL_WEATHER),
("14k-function-calling-cerebras.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14k-function-calling-cerebras.py", EVAL_WEATHER),
("14m-function-calling-openrouter.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14m-function-calling-openrouter.py", EVAL_WEATHER),
("14n-function-calling-perplexity.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14n-function-calling-perplexity.py", EVAL_WEATHER),
("14p-function-calling-gemini-vertex-ai.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14p-function-calling-gemini-vertex-ai.py", EVAL_WEATHER),
("14q-function-calling-qwen.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14q-function-calling-qwen.py", EVAL_WEATHER),
("14r-function-calling-aws.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14r-function-calling-aws.py", EVAL_WEATHER),
("14v-function-calling-openai.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14v-function-calling-openai.py", EVAL_WEATHER),
("14w-function-calling-mistral.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14w-function-calling-mistral.py", EVAL_WEATHER),
("14x-function-calling-openpipe.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("14x-function-calling-openpipe.py", EVAL_WEATHER),
# Video # Video
("14d-function-calling-anthropic-video.py", PROMPT_VISION, EVAL_VISION, BOT_SPEAKS_FIRST), ("14d-function-calling-anthropic-video.py", EVAL_VISION_CAMERA),
("14d-function-calling-aws-video.py", PROMPT_VISION, EVAL_VISION, BOT_SPEAKS_FIRST), ("14d-function-calling-aws-video.py", EVAL_VISION_CAMERA),
("14d-function-calling-gemini-flash-video.py", PROMPT_VISION, EVAL_VISION, BOT_SPEAKS_FIRST), ("14d-function-calling-gemini-flash-video.py", EVAL_VISION_CAMERA),
("14d-function-calling-moondream-video.py", PROMPT_VISION, EVAL_VISION, BOT_SPEAKS_FIRST), ("14d-function-calling-moondream-video.py", EVAL_VISION_CAMERA),
("14d-function-calling-openai-video.py", PROMPT_VISION, EVAL_VISION, BOT_SPEAKS_FIRST), ("14d-function-calling-openai-video.py", EVAL_VISION_CAMERA),
# Currently not working. # Currently not working.
# ("14c-function-calling-together.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), # ("14c-function-calling-together.py", EVAL_WEATHER),
# ("14l-function-calling-deepseek.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), # ("14l-function-calling-deepseek.py", EVAL_WEATHER),
# ("14o-function-calling-gemini-openai-format.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), # ("14o-function-calling-gemini-openai-format.py", EVAL_WEATHER),
] ]
TESTS_15 = [ TESTS_15 = [
("15a-switch-languages.py", PROMPT_SWITCH_LANGUAGE, EVAL_SWITCH_LANGUAGE, BOT_SPEAKS_FIRST), ("15a-switch-languages.py", EVAL_SWITCH_LANGUAGE),
] ]
TESTS_19 = [ TESTS_19 = [
("19-openai-realtime.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("19-openai-realtime.py", EVAL_WEATHER),
("19-openai-realtime-beta.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("19-openai-realtime-beta.py", EVAL_WEATHER),
# OpenAI Realtime not released on Azure yet # OpenAI Realtime not released on Azure yet
# ("19a-azure-realtime.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), # ("19a-azure-realtime.py", EVAL_WEATHER),
("19a-azure-realtime-beta.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("19a-azure-realtime-beta.py", EVAL_WEATHER),
("19b-openai-realtime-text.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("19b-openai-realtime-text.py", EVAL_WEATHER),
("19b-openai-realtime-beta-text.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST), ("19b-openai-realtime-beta-text.py", EVAL_WEATHER),
] ]
TESTS_21 = [ TESTS_21 = [
("21a-tavus-video-service.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("21a-tavus-video-service.py", EVAL_SIMPLE_MATH),
] ]
TESTS_26 = [ TESTS_26 = [
("26-gemini-multimodal-live.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("26-gemini-multimodal-live.py", EVAL_SIMPLE_MATH),
( ("26a-gemini-live-transcription.py", EVAL_SIMPLE_MATH),
"26a-gemini-live-transcription.py", ("26b-gemini-live-function-calling.py", EVAL_WEATHER),
PROMPT_SIMPLE_MATH, ("26c-gemini-live-video.py", EVAL_SIMPLE_MATH),
EVAL_SIMPLE_MATH, ("26e-gemini-multimodal-google-search.py", EVAL_ONLINE_SEARCH),
BOT_SPEAKS_FIRST, ("26h-gemini-live-vertex-function-calling.py", EVAL_WEATHER),
),
(
"26b-gemini-live-function-calling.py",
PROMPT_WEATHER,
EVAL_WEATHER,
BOT_SPEAKS_FIRST,
),
("26c-gemini-live-video.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
(
"26e-gemini-multimodal-google-search.py",
PROMPT_ONLINE_SEARCH,
EVAL_ONLINE_SEARCH,
BOT_SPEAKS_FIRST,
),
# Currently not working. # Currently not working.
# ("26d-gemini-live-text.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), # ("26d-gemini-live-text.py", EVAL_SIMPLE_MATH),
(
"26h-gemini-live-vertex-function-calling.py",
PROMPT_WEATHER,
EVAL_WEATHER,
BOT_SPEAKS_FIRST,
),
] ]
TESTS_27 = [ TESTS_27 = [
("27-simli-layer.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("27-simli-layer.py", EVAL_SIMPLE_MATH),
] ]
TESTS_40 = [ TESTS_40 = [
("40-aws-nova-sonic.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("40-aws-nova-sonic.py", EVAL_SIMPLE_MATH),
] ]
TESTS_43 = [ TESTS_43 = [
("43a-heygen-video-service.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST), ("43a-heygen-video-service.py", EVAL_SIMPLE_MATH),
] ]
TESTS_44 = [ TESTS_44 = [
("44-voicemail-detection.py", PROMPT_VOICEMAIL, EVAL_VOICEMAIL, USER_SPEAKS_FIRST), ("44-voicemail-detection.py", EVAL_VOICEMAIL),
("44-voicemail-detection.py", PROMPT_CONVERSATION, EVAL_CONVERSATION, USER_SPEAKS_FIRST), ("44-voicemail-detection.py", EVAL_CONVERSATION),
] ]
TESTS = [ TESTS = [
*TESTS_07, *TESTS_07,
*TESTS_12,
*TESTS_14, *TESTS_14,
*TESTS_15, *TESTS_15,
*TESTS_19, *TESTS_19,
@@ -240,9 +234,9 @@ async def main(args: argparse.Namespace):
# Parse test config: (test, prompt, eval, user_speaks_first) # Parse test config: (test, prompt, eval, user_speaks_first)
for test_config in TESTS: for test_config in TESTS:
test, prompt, eval, user_speaks_first = test_config test, eval_config = test_config
await runner.run_eval(test, prompt, eval, user_speaks_first) await runner.run_eval(test, eval_config)
runner.print_results() runner.print_results()