scripts(evals): simplify eval configuration and allow RunnerArgs body
This commit is contained in:
@@ -10,9 +10,10 @@ import os
|
|||||||
import re
|
import re
|
||||||
import time
|
import time
|
||||||
import wave
|
import wave
|
||||||
|
from dataclasses import dataclass
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import List, Optional, Tuple
|
from typing import Any, List, Optional, Tuple
|
||||||
|
|
||||||
import aiofiles
|
import aiofiles
|
||||||
from deepgram import LiveOptions
|
from deepgram import LiveOptions
|
||||||
@@ -53,6 +54,14 @@ EVAL_TIMEOUT_SECS = 120
|
|||||||
EvalPrompt = str | Tuple[str, ImageFile]
|
EvalPrompt = str | Tuple[str, ImageFile]
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class EvalConfig:
|
||||||
|
prompt: EvalPrompt
|
||||||
|
eval: str
|
||||||
|
eval_speaks_first: bool = False
|
||||||
|
runner_args_body: Optional[Any] = None
|
||||||
|
|
||||||
|
|
||||||
class EvalRunner:
|
class EvalRunner:
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
@@ -93,9 +102,7 @@ class EvalRunner:
|
|||||||
async def run_eval(
|
async def run_eval(
|
||||||
self,
|
self,
|
||||||
example_file: str,
|
example_file: str,
|
||||||
prompt: EvalPrompt,
|
eval_config: EvalConfig,
|
||||||
eval: str,
|
|
||||||
user_speaks_first: bool = False,
|
|
||||||
):
|
):
|
||||||
if not re.match(self._pattern, example_file):
|
if not re.match(self._pattern, example_file):
|
||||||
return
|
return
|
||||||
@@ -112,10 +119,8 @@ class EvalRunner:
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
tasks = [
|
tasks = [
|
||||||
asyncio.create_task(run_example_pipeline(script_path)),
|
asyncio.create_task(run_example_pipeline(script_path, eval_config)),
|
||||||
asyncio.create_task(
|
asyncio.create_task(run_eval_pipeline(self, example_file, eval_config)),
|
||||||
run_eval_pipeline(self, example_file, prompt, eval, user_speaks_first)
|
|
||||||
),
|
|
||||||
]
|
]
|
||||||
_, pending = await asyncio.wait(tasks, timeout=EVAL_TIMEOUT_SECS)
|
_, pending = await asyncio.wait(tasks, timeout=EVAL_TIMEOUT_SECS)
|
||||||
if pending:
|
if pending:
|
||||||
@@ -177,7 +182,7 @@ class EvalRunner:
|
|||||||
return os.path.join(self._recordings_dir, f"{base_name}.wav")
|
return os.path.join(self._recordings_dir, f"{base_name}.wav")
|
||||||
|
|
||||||
|
|
||||||
async def run_example_pipeline(script_path: Path):
|
async def run_example_pipeline(script_path: Path, eval_config: EvalConfig):
|
||||||
room_url = os.getenv("DAILY_SAMPLE_ROOM_URL")
|
room_url = os.getenv("DAILY_SAMPLE_ROOM_URL")
|
||||||
|
|
||||||
module = load_module_from_path(script_path)
|
module = load_module_from_path(script_path)
|
||||||
@@ -196,6 +201,7 @@ async def run_example_pipeline(script_path: Path):
|
|||||||
|
|
||||||
runner_args = RunnerArguments()
|
runner_args = RunnerArguments()
|
||||||
runner_args.pipeline_idle_timeout_secs = PIPELINE_IDLE_TIMEOUT_SECS
|
runner_args.pipeline_idle_timeout_secs = PIPELINE_IDLE_TIMEOUT_SECS
|
||||||
|
runner_args.body = eval_config.runner_args_body
|
||||||
|
|
||||||
await module.run_bot(transport, runner_args)
|
await module.run_bot(transport, runner_args)
|
||||||
|
|
||||||
@@ -203,9 +209,7 @@ async def run_example_pipeline(script_path: Path):
|
|||||||
async def run_eval_pipeline(
|
async def run_eval_pipeline(
|
||||||
eval_runner: EvalRunner,
|
eval_runner: EvalRunner,
|
||||||
example_file: str,
|
example_file: str,
|
||||||
prompt: EvalPrompt,
|
eval_config: EvalConfig,
|
||||||
eval: str,
|
|
||||||
user_speaks_first: bool = False,
|
|
||||||
):
|
):
|
||||||
logger.info(f"Starting eval bot")
|
logger.info(f"Starting eval bot")
|
||||||
|
|
||||||
@@ -262,17 +266,17 @@ async def run_eval_pipeline(
|
|||||||
# Load example prompt depending on image.
|
# Load example prompt depending on image.
|
||||||
example_prompt = ""
|
example_prompt = ""
|
||||||
example_image: Optional[ImageFile] = None
|
example_image: Optional[ImageFile] = None
|
||||||
if isinstance(prompt, str):
|
if isinstance(eval_config.prompt, str):
|
||||||
example_prompt = prompt
|
example_prompt = eval_config.prompt
|
||||||
elif isinstance(prompt, tuple):
|
elif isinstance(eval_config.prompt, tuple):
|
||||||
example_prompt, example_image = prompt
|
example_prompt, example_image = eval_config.prompt
|
||||||
|
|
||||||
eval_prompt = f"The answer is correct if it matches: {eval}."
|
eval_prompt = f"The answer is correct if it matches: {eval}."
|
||||||
common_system_prompt = (
|
common_system_prompt = (
|
||||||
"The user might say things other than the answer and that's allowed. "
|
"The user might say things other than the answer and that's allowed. "
|
||||||
f"You should only call the eval function with your assessment when the user actually answers the question. {eval_prompt}"
|
f"You should only call the eval function with your assessment when the user actually answers the question. {eval_prompt}"
|
||||||
)
|
)
|
||||||
if user_speaks_first:
|
if eval_config.eval_speaks_first:
|
||||||
system_prompt = f"You are an LLM eval, be extremly brief. You will start the conversation by saying: '{example_prompt}'. {common_system_prompt}"
|
system_prompt = f"You are an LLM eval, be extremly brief. You will start the conversation by saying: '{example_prompt}'. {common_system_prompt}"
|
||||||
else:
|
else:
|
||||||
system_prompt = f"You are an LLM eval, be extremly brief. Your goal is to first ask one question: {example_prompt}. {common_system_prompt}"
|
system_prompt = f"You are an LLM eval, be extremly brief. Your goal is to first ask one question: {example_prompt}. {common_system_prompt}"
|
||||||
@@ -330,9 +334,9 @@ async def run_eval_pipeline(
|
|||||||
|
|
||||||
# Default behavior is for the bot to speak first
|
# Default behavior is for the bot to speak first
|
||||||
# If the eval bot speaks first, we append the prompt to the messages
|
# If the eval bot speaks first, we append the prompt to the messages
|
||||||
if user_speaks_first:
|
if eval_config.eval_speaks_first:
|
||||||
messages.append(
|
messages.append(
|
||||||
{"role": "user", "content": f"Start by saying this exactly: '{prompt}'"}
|
{"role": "user", "content": f"Start by saying this exactly: '{eval_config.prompt}'"}
|
||||||
)
|
)
|
||||||
await task.queue_frames([LLMRunFrame()])
|
await task.queue_frames([LLMRunFrame()])
|
||||||
|
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ from datetime import datetime, timezone
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from eval import EvalRunner
|
from eval import EvalConfig, EvalRunner
|
||||||
from loguru import logger
|
from loguru import logger
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
from utils import check_env_variables
|
from utils import check_env_variables
|
||||||
@@ -24,190 +24,184 @@ ASSETS_DIR = SCRIPT_DIR / "assets"
|
|||||||
|
|
||||||
FOUNDATIONAL_DIR = SCRIPT_DIR.parent.parent / "examples" / "foundational"
|
FOUNDATIONAL_DIR = SCRIPT_DIR.parent.parent / "examples" / "foundational"
|
||||||
|
|
||||||
# Speaking order constants
|
EVAL_SIMPLE_MATH = EvalConfig(
|
||||||
USER_SPEAKS_FIRST = True
|
prompt="A simple math addition.",
|
||||||
BOT_SPEAKS_FIRST = False
|
eval="Correct math addition.",
|
||||||
|
|
||||||
# Math
|
|
||||||
PROMPT_SIMPLE_MATH = "A simple math addition."
|
|
||||||
EVAL_SIMPLE_MATH = "Correct math addition."
|
|
||||||
|
|
||||||
# Weather
|
|
||||||
PROMPT_WEATHER = "What's the weather in San Francisco?"
|
|
||||||
EVAL_WEATHER = (
|
|
||||||
"Something specific about the current weather in San Francisco, including the degrees."
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# Online search
|
EVAL_WEATHER = EvalConfig(
|
||||||
PROMPT_ONLINE_SEARCH = "What's the date right now in London?"
|
prompt="What's the weather in San Francisco?",
|
||||||
EVAL_ONLINE_SEARCH = f"Today is {datetime.now(timezone.utc).strftime('%B %d, %Y')}."
|
eval="Something specific about the current weather in San Francisco, including the degrees.",
|
||||||
|
)
|
||||||
|
|
||||||
# Switch language
|
EVAL_ONLINE_SEARCH = EvalConfig(
|
||||||
PROMPT_SWITCH_LANGUAGE = "Say something in Spanish."
|
prompt="What's the date right now in London?",
|
||||||
EVAL_SWITCH_LANGUAGE = "The user is now talking in Spanish."
|
eval=f"Today is {datetime.now(timezone.utc).strftime('%B %d, %Y')}.",
|
||||||
|
)
|
||||||
|
|
||||||
# Vision
|
EVAL_SWITCH_LANGUAGE = EvalConfig(
|
||||||
PROMPT_VISION = ("Briefly describe what you see.", Image.open(ASSETS_DIR / "cat.jpg"))
|
prompt="Say something in Spanish.",
|
||||||
EVAL_VISION = "A cat description."
|
eval="The user is now talking in Spanish.",
|
||||||
|
)
|
||||||
|
|
||||||
|
EVAL_VISION_CAMERA = EvalConfig(
|
||||||
|
prompt=("Briefly describe what you see.", Image.open(ASSETS_DIR / "cat.jpg")),
|
||||||
|
eval="A cat description.",
|
||||||
|
)
|
||||||
|
|
||||||
|
EVAL_VISION_IMAGE = EvalConfig(
|
||||||
|
prompt="Briefly describe this image.",
|
||||||
|
eval="A cat description.",
|
||||||
|
eval_speaks_first=True,
|
||||||
|
runner_args_body={
|
||||||
|
"image_path": ASSETS_DIR / "cat.jpg",
|
||||||
|
"question": "Briefly describe this image.",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
EVAL_VOICEMAIL = EvalConfig(
|
||||||
|
prompt="Please leave a message after the beep.",
|
||||||
|
eval="Assess the conversation and determine if it is a voicemail.",
|
||||||
|
eval_speaks_first=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
EVAL_CONVERSATION = EvalConfig(
|
||||||
|
prompt="Hello, this is Mark.",
|
||||||
|
eval="A start of a conversation, not a voicemail.",
|
||||||
|
eval_speaks_first=True,
|
||||||
|
)
|
||||||
|
|
||||||
# Voicemail
|
|
||||||
PROMPT_VOICEMAIL = "Please leave a message after the beep."
|
|
||||||
EVAL_VOICEMAIL = "Assess the conversation and determine if it is a voicemail."
|
|
||||||
PROMPT_CONVERSATION = "Hello, this is Mark."
|
|
||||||
EVAL_CONVERSATION = "A start of a conversation, not a voicemail."
|
|
||||||
|
|
||||||
TESTS_07 = [
|
TESTS_07 = [
|
||||||
# 07 series
|
# 07 series
|
||||||
("07-interruptible.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07-interruptible.py", EVAL_SIMPLE_MATH),
|
||||||
("07-interruptible-cartesia-http.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07-interruptible-cartesia-http.py", EVAL_SIMPLE_MATH),
|
||||||
("07a-interruptible-speechmatics.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07a-interruptible-speechmatics.py", EVAL_SIMPLE_MATH),
|
||||||
("07aa-interruptible-soniox.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07aa-interruptible-soniox.py", EVAL_SIMPLE_MATH),
|
||||||
("07ab-interruptible-inworld-http.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07ab-interruptible-inworld-http.py", EVAL_SIMPLE_MATH),
|
||||||
("07ac-interruptible-asyncai.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07ac-interruptible-asyncai.py", EVAL_SIMPLE_MATH),
|
||||||
("07ac-interruptible-asyncai-http.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07ac-interruptible-asyncai-http.py", EVAL_SIMPLE_MATH),
|
||||||
("07b-interruptible-langchain.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07b-interruptible-langchain.py", EVAL_SIMPLE_MATH),
|
||||||
("07c-interruptible-deepgram.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07c-interruptible-deepgram.py", EVAL_SIMPLE_MATH),
|
||||||
("07c-interruptible-deepgram-flux.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07c-interruptible-deepgram-flux.py", EVAL_SIMPLE_MATH),
|
||||||
("07d-interruptible-elevenlabs.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07d-interruptible-elevenlabs.py", EVAL_SIMPLE_MATH),
|
||||||
(
|
("07d-interruptible-elevenlabs-http.py", EVAL_SIMPLE_MATH),
|
||||||
"07d-interruptible-elevenlabs-http.py",
|
("07f-interruptible-azure.py", EVAL_SIMPLE_MATH),
|
||||||
PROMPT_SIMPLE_MATH,
|
("07g-interruptible-openai.py", EVAL_SIMPLE_MATH),
|
||||||
EVAL_SIMPLE_MATH,
|
("07h-interruptible-openpipe.py", EVAL_SIMPLE_MATH),
|
||||||
BOT_SPEAKS_FIRST,
|
("07j-interruptible-gladia.py", EVAL_SIMPLE_MATH),
|
||||||
),
|
("07k-interruptible-lmnt.py", EVAL_SIMPLE_MATH),
|
||||||
("07f-interruptible-azure.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07l-interruptible-groq.py", EVAL_SIMPLE_MATH),
|
||||||
("07g-interruptible-openai.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07m-interruptible-aws.py", EVAL_SIMPLE_MATH),
|
||||||
("07h-interruptible-openpipe.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07m-interruptible-aws-strands.py", EVAL_WEATHER),
|
||||||
("07j-interruptible-gladia.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07n-interruptible-gemini.py", EVAL_SIMPLE_MATH),
|
||||||
("07k-interruptible-lmnt.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07n-interruptible-google.py", EVAL_SIMPLE_MATH),
|
||||||
("07l-interruptible-groq.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07o-interruptible-assemblyai.py", EVAL_SIMPLE_MATH),
|
||||||
("07m-interruptible-aws.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07q-interruptible-rime.py", EVAL_SIMPLE_MATH),
|
||||||
("07m-interruptible-aws-strands.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("07q-interruptible-rime-http.py", EVAL_SIMPLE_MATH),
|
||||||
("07n-interruptible-gemini.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07r-interruptible-riva-nim.py", EVAL_SIMPLE_MATH),
|
||||||
("07n-interruptible-google.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07s-interruptible-google-audio-in.py", EVAL_SIMPLE_MATH),
|
||||||
("07o-interruptible-assemblyai.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07t-interruptible-fish.py", EVAL_SIMPLE_MATH),
|
||||||
("07q-interruptible-rime.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07v-interruptible-neuphonic.py", EVAL_SIMPLE_MATH),
|
||||||
("07q-interruptible-rime-http.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07v-interruptible-neuphonic-http.py", EVAL_SIMPLE_MATH),
|
||||||
("07r-interruptible-riva-nim.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("07w-interruptible-fal.py", EVAL_SIMPLE_MATH),
|
||||||
(
|
("07y-interruptible-minimax.py", EVAL_SIMPLE_MATH),
|
||||||
"07s-interruptible-google-audio-in.py",
|
("07z-interruptible-sarvam.py", EVAL_SIMPLE_MATH),
|
||||||
PROMPT_SIMPLE_MATH,
|
("07ae-interruptible-hume.py", EVAL_SIMPLE_MATH),
|
||||||
EVAL_SIMPLE_MATH,
|
|
||||||
BOT_SPEAKS_FIRST,
|
|
||||||
),
|
|
||||||
("07t-interruptible-fish.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
|
||||||
("07v-interruptible-neuphonic.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
|
||||||
("07v-interruptible-neuphonic-http.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
|
||||||
("07w-interruptible-fal.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
|
||||||
("07y-interruptible-minimax.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
|
||||||
("07z-interruptible-sarvam.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
|
||||||
("07ae-interruptible-hume.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
|
||||||
# Needs a local XTTS docker instance running.
|
# Needs a local XTTS docker instance running.
|
||||||
# ("07i-interruptible-xtts.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
# ("07i-interruptible-xtts.py", EVAL_SIMPLE_MATH),
|
||||||
# Needs a Krisp license.
|
# Needs a Krisp license.
|
||||||
# ("07p-interruptible-krisp.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
# ("07p-interruptible-krisp.py", EVAL_SIMPLE_MATH),
|
||||||
# Needs GPU resources.
|
# Needs GPU resources.
|
||||||
# ("07u-interruptible-ultravox.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
# ("07u-interruptible-ultravox.py", EVAL_SIMPLE_MATH),
|
||||||
|
]
|
||||||
|
|
||||||
|
TESTS_12 = [
|
||||||
|
("12-describe-image-openai.py", EVAL_VISION_IMAGE),
|
||||||
|
("12a-describe-image-anthropic.py", EVAL_VISION_IMAGE),
|
||||||
|
("12b-describe-image-aws.py", EVAL_VISION_IMAGE),
|
||||||
|
("12c-describe-image-gemini-flash.py", EVAL_VISION_IMAGE),
|
||||||
]
|
]
|
||||||
|
|
||||||
TESTS_14 = [
|
TESTS_14 = [
|
||||||
("14-function-calling.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14-function-calling.py", EVAL_WEATHER),
|
||||||
("14a-function-calling-anthropic.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14a-function-calling-anthropic.py", EVAL_WEATHER),
|
||||||
("14e-function-calling-google.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14e-function-calling-google.py", EVAL_WEATHER),
|
||||||
("14f-function-calling-groq.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14f-function-calling-groq.py", EVAL_WEATHER),
|
||||||
("14g-function-calling-grok.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14g-function-calling-grok.py", EVAL_WEATHER),
|
||||||
("14h-function-calling-azure.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14h-function-calling-azure.py", EVAL_WEATHER),
|
||||||
("14i-function-calling-fireworks.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14i-function-calling-fireworks.py", EVAL_WEATHER),
|
||||||
("14j-function-calling-nim.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14j-function-calling-nim.py", EVAL_WEATHER),
|
||||||
("14k-function-calling-cerebras.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14k-function-calling-cerebras.py", EVAL_WEATHER),
|
||||||
("14m-function-calling-openrouter.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14m-function-calling-openrouter.py", EVAL_WEATHER),
|
||||||
("14n-function-calling-perplexity.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14n-function-calling-perplexity.py", EVAL_WEATHER),
|
||||||
("14p-function-calling-gemini-vertex-ai.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14p-function-calling-gemini-vertex-ai.py", EVAL_WEATHER),
|
||||||
("14q-function-calling-qwen.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14q-function-calling-qwen.py", EVAL_WEATHER),
|
||||||
("14r-function-calling-aws.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14r-function-calling-aws.py", EVAL_WEATHER),
|
||||||
("14v-function-calling-openai.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14v-function-calling-openai.py", EVAL_WEATHER),
|
||||||
("14w-function-calling-mistral.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14w-function-calling-mistral.py", EVAL_WEATHER),
|
||||||
("14x-function-calling-openpipe.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("14x-function-calling-openpipe.py", EVAL_WEATHER),
|
||||||
# Video
|
# Video
|
||||||
("14d-function-calling-anthropic-video.py", PROMPT_VISION, EVAL_VISION, BOT_SPEAKS_FIRST),
|
("14d-function-calling-anthropic-video.py", EVAL_VISION_CAMERA),
|
||||||
("14d-function-calling-aws-video.py", PROMPT_VISION, EVAL_VISION, BOT_SPEAKS_FIRST),
|
("14d-function-calling-aws-video.py", EVAL_VISION_CAMERA),
|
||||||
("14d-function-calling-gemini-flash-video.py", PROMPT_VISION, EVAL_VISION, BOT_SPEAKS_FIRST),
|
("14d-function-calling-gemini-flash-video.py", EVAL_VISION_CAMERA),
|
||||||
("14d-function-calling-moondream-video.py", PROMPT_VISION, EVAL_VISION, BOT_SPEAKS_FIRST),
|
("14d-function-calling-moondream-video.py", EVAL_VISION_CAMERA),
|
||||||
("14d-function-calling-openai-video.py", PROMPT_VISION, EVAL_VISION, BOT_SPEAKS_FIRST),
|
("14d-function-calling-openai-video.py", EVAL_VISION_CAMERA),
|
||||||
# Currently not working.
|
# Currently not working.
|
||||||
# ("14c-function-calling-together.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
# ("14c-function-calling-together.py", EVAL_WEATHER),
|
||||||
# ("14l-function-calling-deepseek.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
# ("14l-function-calling-deepseek.py", EVAL_WEATHER),
|
||||||
# ("14o-function-calling-gemini-openai-format.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
# ("14o-function-calling-gemini-openai-format.py", EVAL_WEATHER),
|
||||||
]
|
]
|
||||||
|
|
||||||
TESTS_15 = [
|
TESTS_15 = [
|
||||||
("15a-switch-languages.py", PROMPT_SWITCH_LANGUAGE, EVAL_SWITCH_LANGUAGE, BOT_SPEAKS_FIRST),
|
("15a-switch-languages.py", EVAL_SWITCH_LANGUAGE),
|
||||||
]
|
]
|
||||||
|
|
||||||
TESTS_19 = [
|
TESTS_19 = [
|
||||||
("19-openai-realtime.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("19-openai-realtime.py", EVAL_WEATHER),
|
||||||
("19-openai-realtime-beta.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("19-openai-realtime-beta.py", EVAL_WEATHER),
|
||||||
# OpenAI Realtime not released on Azure yet
|
# OpenAI Realtime not released on Azure yet
|
||||||
# ("19a-azure-realtime.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
# ("19a-azure-realtime.py", EVAL_WEATHER),
|
||||||
("19a-azure-realtime-beta.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("19a-azure-realtime-beta.py", EVAL_WEATHER),
|
||||||
("19b-openai-realtime-text.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("19b-openai-realtime-text.py", EVAL_WEATHER),
|
||||||
("19b-openai-realtime-beta-text.py", PROMPT_WEATHER, EVAL_WEATHER, BOT_SPEAKS_FIRST),
|
("19b-openai-realtime-beta-text.py", EVAL_WEATHER),
|
||||||
]
|
]
|
||||||
|
|
||||||
TESTS_21 = [
|
TESTS_21 = [
|
||||||
("21a-tavus-video-service.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("21a-tavus-video-service.py", EVAL_SIMPLE_MATH),
|
||||||
]
|
]
|
||||||
|
|
||||||
TESTS_26 = [
|
TESTS_26 = [
|
||||||
("26-gemini-multimodal-live.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("26-gemini-multimodal-live.py", EVAL_SIMPLE_MATH),
|
||||||
(
|
("26a-gemini-live-transcription.py", EVAL_SIMPLE_MATH),
|
||||||
"26a-gemini-live-transcription.py",
|
("26b-gemini-live-function-calling.py", EVAL_WEATHER),
|
||||||
PROMPT_SIMPLE_MATH,
|
("26c-gemini-live-video.py", EVAL_SIMPLE_MATH),
|
||||||
EVAL_SIMPLE_MATH,
|
("26e-gemini-multimodal-google-search.py", EVAL_ONLINE_SEARCH),
|
||||||
BOT_SPEAKS_FIRST,
|
("26h-gemini-live-vertex-function-calling.py", EVAL_WEATHER),
|
||||||
),
|
|
||||||
(
|
|
||||||
"26b-gemini-live-function-calling.py",
|
|
||||||
PROMPT_WEATHER,
|
|
||||||
EVAL_WEATHER,
|
|
||||||
BOT_SPEAKS_FIRST,
|
|
||||||
),
|
|
||||||
("26c-gemini-live-video.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
|
||||||
(
|
|
||||||
"26e-gemini-multimodal-google-search.py",
|
|
||||||
PROMPT_ONLINE_SEARCH,
|
|
||||||
EVAL_ONLINE_SEARCH,
|
|
||||||
BOT_SPEAKS_FIRST,
|
|
||||||
),
|
|
||||||
# Currently not working.
|
# Currently not working.
|
||||||
# ("26d-gemini-live-text.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
# ("26d-gemini-live-text.py", EVAL_SIMPLE_MATH),
|
||||||
(
|
|
||||||
"26h-gemini-live-vertex-function-calling.py",
|
|
||||||
PROMPT_WEATHER,
|
|
||||||
EVAL_WEATHER,
|
|
||||||
BOT_SPEAKS_FIRST,
|
|
||||||
),
|
|
||||||
]
|
]
|
||||||
|
|
||||||
TESTS_27 = [
|
TESTS_27 = [
|
||||||
("27-simli-layer.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("27-simli-layer.py", EVAL_SIMPLE_MATH),
|
||||||
]
|
]
|
||||||
|
|
||||||
TESTS_40 = [
|
TESTS_40 = [
|
||||||
("40-aws-nova-sonic.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("40-aws-nova-sonic.py", EVAL_SIMPLE_MATH),
|
||||||
]
|
]
|
||||||
|
|
||||||
TESTS_43 = [
|
TESTS_43 = [
|
||||||
("43a-heygen-video-service.py", PROMPT_SIMPLE_MATH, EVAL_SIMPLE_MATH, BOT_SPEAKS_FIRST),
|
("43a-heygen-video-service.py", EVAL_SIMPLE_MATH),
|
||||||
]
|
]
|
||||||
|
|
||||||
TESTS_44 = [
|
TESTS_44 = [
|
||||||
("44-voicemail-detection.py", PROMPT_VOICEMAIL, EVAL_VOICEMAIL, USER_SPEAKS_FIRST),
|
("44-voicemail-detection.py", EVAL_VOICEMAIL),
|
||||||
("44-voicemail-detection.py", PROMPT_CONVERSATION, EVAL_CONVERSATION, USER_SPEAKS_FIRST),
|
("44-voicemail-detection.py", EVAL_CONVERSATION),
|
||||||
]
|
]
|
||||||
|
|
||||||
TESTS = [
|
TESTS = [
|
||||||
*TESTS_07,
|
*TESTS_07,
|
||||||
|
*TESTS_12,
|
||||||
*TESTS_14,
|
*TESTS_14,
|
||||||
*TESTS_15,
|
*TESTS_15,
|
||||||
*TESTS_19,
|
*TESTS_19,
|
||||||
@@ -240,9 +234,9 @@ async def main(args: argparse.Namespace):
|
|||||||
|
|
||||||
# Parse test config: (test, prompt, eval, user_speaks_first)
|
# Parse test config: (test, prompt, eval, user_speaks_first)
|
||||||
for test_config in TESTS:
|
for test_config in TESTS:
|
||||||
test, prompt, eval, user_speaks_first = test_config
|
test, eval_config = test_config
|
||||||
|
|
||||||
await runner.run_eval(test, prompt, eval, user_speaks_first)
|
await runner.run_eval(test, eval_config)
|
||||||
|
|
||||||
runner.print_results()
|
runner.print_results()
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user