diff --git a/README.md b/README.md index 7b2a179c..2586b118 100644 --- a/README.md +++ b/README.md @@ -334,6 +334,12 @@ Eval( - Translation - Fine-tuned binary classifiers +### Voice evaluations + +- Speech clarity +- Turn-taking +- Voice task success + ### RAG evaluations - Context precision diff --git a/SCORERS.md b/SCORERS.md index 56893245..ca9b133a 100644 --- a/SCORERS.md +++ b/SCORERS.md @@ -194,6 +194,57 @@ Evaluates translation quality. --- +## Voice scorers + +`SpeechClarity` and `TurnTaking` listen to a voice call recording, not a transcript. They default to `gpt-audio`, which accepts WAV and MP3 audio. `VoiceTaskSuccess` reads the call's transcript from the trace. + +### SpeechClarity + +Evaluates how easily a listener on a phone call can understand every word the agent says: garbled or cut-off words, distortion, dropouts, echo, and background noise. + +**Parameters:** + +- `audio` (object, required): The recording as OpenAI `input_audio`: base64 `data` and a `format` of `"wav"` or `"mp3"` +- `model` (string, optional): An audio-capable model to use + +**Score Range:** 0-1 + +- `1.0` = Clear and understandable without effort +- `0.5` = Audible flaws, but easy to understand +- `0.0` = Important words are unclear or lost + +### TurnTaking + +Evaluates how the agent handles turn-taking: talking over the caller, cutting them off, or not stopping when interrupted. Response latency is ignored. + +**Parameters:** + +- `audio` (object, required): The recording as OpenAI `input_audio`: base64 `data` and a `format` of `"wav"` or `"mp3"` +- `model` (string, optional): An audio-capable model to use + +**Score Range:** 0-1 + +- `1.0` = Smooth turn-taking +- `0.5` = Brief overlaps or a slow yield, but no caller words are lost +- `0.0` = Caller words are lost + +### VoiceTaskSuccess + +Evaluates whether the agent correctly completed the caller's request, from the trace's `{{thread_with_system}}`: its instructions, tool calls, tool results, and replies. + +**Parameters:** + +- `trace` (Trace, required): The voice call's trace +- `model` (string, optional): Model to use + +**Score Range:** 0-1 + +- `1.0` = Completed correctly, and everything told to the caller is supported +- `0.5` = Completed, with minor problems that did not change the outcome +- `0.0` = Not completed, a wrong or unauthorized action, or false or unsupported claims + +--- + ## RAG (Retrieval-Augmented Generation) scorers These scorers evaluate RAG systems by assessing both context retrieval and answer generation quality. diff --git a/js/llm.test.ts b/js/llm.test.ts index 5863eecc..62ff480b 100644 --- a/js/llm.test.ts +++ b/js/llm.test.ts @@ -8,6 +8,7 @@ import { buildClassificationTools, LLMClassifierFromTemplate, OpenAIClassifier, + SpeechClarity, templateUsesThreadVariables, } from "../js/llm"; import { @@ -74,6 +75,63 @@ describe("LLM Tests", () => { ).toBe(true); }); + test("SpeechClarity sends the audio to an audio model", async () => { + let body: any; + server.use( + http.post( + "https://api.openai.com/v1/chat/completions", + async ({ request }) => { + body = await request.json(); + return HttpResponse.json({ + id: "chatcmpl-test", + object: "chat.completion", + created: 0, + model: body.model, + choices: [ + { + index: 0, + finish_reason: "tool_calls", + message: { + role: "assistant", + content: null, + tool_calls: [ + { + id: "call_test", + type: "function", + function: { + name: "select_choice", + arguments: JSON.stringify({ + reasons: "Clear.", + choice: "A", + }), + }, + }, + ], + }, + }, + ], + }); + }, + ), + ); + + const audio = { data: "UklGRg==", format: "wav" as const }; + const score = await SpeechClarity({ output: "", audio }); + + expect(score.score).toBe(1); + expect(body.model).toBe("gpt-audio"); + expect(body.messages.at(-1)).toEqual({ + role: "user", + content: [{ type: "input_audio", input_audio: audio }], + }); + }); + + test("SpeechClarity requires audio", async () => { + await expect(SpeechClarity({ output: "" } as any)).rejects.toThrow( + "SpeechClarity needs the call recording", + ); + }); + test("openai classifier should evaluate titles", async () => { let callCount = -1; server.use( diff --git a/js/llm.ts b/js/llm.ts index 1b7f318e..889ab1af 100644 --- a/js/llm.ts +++ b/js/llm.ts @@ -7,6 +7,7 @@ import { } from "./oai"; import { ModelGradedSpec, templates } from "./templates"; import { + ChatCompletionContentPartInputAudio, ChatCompletionMessage, ChatCompletionMessageParam, ChatCompletionTool, @@ -143,6 +144,7 @@ export type OpenAIClassifierArgs = { choiceScores: Record; classificationTools: ChatCompletionTool[]; cache?: ChatCache; + audio?: ChatCompletionContentPartInputAudio.InputAudio; } & LLMArgs & RenderArgs; @@ -175,6 +177,7 @@ export async function OpenAIClassifier( reasoningBudget, useResponsesApi, cache, + audio, ...remainingRenderArgs } = remaining; @@ -212,6 +215,12 @@ export async function OpenAIClassifier( }; const messages = renderMessages(messagesArg, renderArgs); + if (audio) { + messages.push({ + role: "user", + content: [{ type: "input_audio", input_audio: audio }], + }); + } const resp = await cachedChatCompletion( { @@ -306,6 +315,7 @@ export function LLMClassifierFromTemplate({ reasoningEnabled, reasoningBudget, useResponsesApi, + requiresAudio, }: { name: string; promptTemplate: string; @@ -318,6 +328,7 @@ export function LLMClassifierFromTemplate({ reasoningEnabled?: boolean; reasoningBudget?: number; useResponsesApi?: boolean; + requiresAudio?: boolean; }): Scorer> { const choiceStrings = Object.keys(choiceScores); const ret = async ( @@ -376,6 +387,10 @@ export function LLMClassifierFromTemplate({ useCoT, }; + if (requiresAudio && !classifierArgs.audio) { + throw new Error(`${name} needs the call recording as \`audio\``); + } + return await OpenAIClassifier(classifierArgs); }; Object.defineProperty(ret, "name", { @@ -398,6 +413,7 @@ export function LLMClassifierFromSpec( useCoT: spec.use_cot, temperature: spec.temperature, maxTokens: spec.max_tokens, + requiresAudio: spec.requires_audio, }); } @@ -492,3 +508,25 @@ export const Translation = buildLLMClassifier<{ language: string; input: string; }>("Translation", "translation"); + +/** + * Test whether the agent in a voice call `audio` recording speaks clearly. + */ +export const SpeechClarity = buildLLMClassifier<{ + audio: ChatCompletionContentPartInputAudio.InputAudio; +}>("SpeechClarity", "speech_clarity"); + +/** + * Test whether the agent in a voice call `audio` recording takes turns smoothly. + */ +export const TurnTaking = buildLLMClassifier<{ + audio: ChatCompletionContentPartInputAudio.InputAudio; +}>("TurnTaking", "turn_taking"); + +/** + * Test whether a voice agent completed the caller's request, from the trace's thread. + */ +export const VoiceTaskSuccess = buildLLMClassifier<{}>( + "VoiceTaskSuccess", + "voice_task_success", +); diff --git a/js/manifest.ts b/js/manifest.ts index c21d3703..6dd06587 100644 --- a/js/manifest.ts +++ b/js/manifest.ts @@ -6,9 +6,12 @@ import { Humor, Possible, Security, + SpeechClarity, Sql, Summary, Translation, + TurnTaking, + VoiceTaskSuccess, } from "./llm"; import { NumericDiff } from "./number"; import { EmbeddingSimilarity, Levenshtein } from "./string"; @@ -103,6 +106,29 @@ export const Evaluators: { }, ], }, + { + label: "Voice", + methods: [ + { + method: SpeechClarity, + description: + "Test whether the agent in a voice call recording speaks clearly.", + template: templates.speech_clarity, + }, + { + method: TurnTaking, + description: + "Test whether the agent in a voice call recording takes turns smoothly.", + template: templates.turn_taking, + }, + { + method: VoiceTaskSuccess, + description: + "Test whether a voice agent correctly completed the caller's request, from the trace's thread.", + template: templates.voice_task_success, + }, + ], + }, { label: "RAG", methods: [ diff --git a/js/templates.ts b/js/templates.ts index 69f4637c..ef661404 100644 --- a/js/templates.ts +++ b/js/templates.ts @@ -7,9 +7,12 @@ import factuality from "../templates/factuality.yaml"; import humor from "../templates/humor.yaml"; import possible from "../templates/possible.yaml"; import security from "../templates/security.yaml"; +import speech_clarity from "../templates/speech_clarity.yaml"; import sql from "../templates/sql.yaml"; import summary from "../templates/summary.yaml"; import translation from "../templates/translation.yaml"; +import turn_taking from "../templates/turn_taking.yaml"; +import voice_task_success from "../templates/voice_task_success.yaml"; export const modelGradedSpecSchema = z.object({ prompt: z.string(), @@ -18,6 +21,7 @@ export const modelGradedSpecSchema = z.object({ use_cot: z.boolean().optional(), temperature: z.number().optional(), max_tokens: z.number().optional(), + requires_audio: z.boolean().optional(), }); export type ModelGradedSpec = z.infer; @@ -29,9 +33,12 @@ const templateStrings = { humor, possible, security, + speech_clarity, sql, summary, translation, + turn_taking, + voice_task_success, } as const; // eslint-disable-next-line @typescript-eslint/consistent-type-assertions diff --git a/py/autoevals/llm.py b/py/autoevals/llm.py index b0093cb4..d23cc8c8 100644 --- a/py/autoevals/llm.py +++ b/py/autoevals/llm.py @@ -225,9 +225,12 @@ def _name(self): return self.name def _build_args(self, output, expected, **kwargs): + messages = self._render_messages(output=output, expected=expected, **kwargs) + if kwargs.get("audio"): + messages.append({"role": "user", "content": [{"type": "input_audio", "input_audio": kwargs["audio"]}]}) return dict( model=self.model, - messages=self._render_messages(output=output, expected=expected, **kwargs), + messages=messages, tools=self.classification_tools, tool_choice={"type": "function", "function": {"name": "select_choice"}}, ) @@ -301,6 +304,7 @@ class ModelGradedSpec: use_cot: bool | None = None temperature: float | None = None max_tokens: int | None = None + requires_audio: bool | None = None class LLMClassifier(OpenAILLMClassifier): @@ -375,9 +379,11 @@ def __init__( api_key=None, base_url=None, client: Client | None = None, + requires_audio=False, **extra_render_args, ): self._template_uses_thread_variables = template_uses_thread_variables(prompt_template) + self._requires_audio = requires_audio choice_strings = list(choice_scores.keys()) # Use configured default model if not specified if model is None: @@ -410,6 +416,11 @@ def __init__( client=client, ) + def _build_args(self, output, expected, **kwargs): + if self._requires_audio and not kwargs.get("audio"): + raise ValueError(f"{self.name} needs the call recording as `audio`") + return super()._build_args(output, expected, **kwargs) + @staticmethod def _get_trace_thread_method(trace) -> Callable[..., object] | None: if hasattr(trace, "get_thread") and callable(trace.get_thread): @@ -484,6 +495,8 @@ def from_spec(cls, name: str, spec: ModelGradedSpec, client: Client | None = Non spec_kwargs["temperature"] = spec.temperature if spec.max_tokens is not None: spec_kwargs["max_tokens"] = spec.max_tokens + if spec.requires_audio is not None: + spec_kwargs["requires_audio"] = spec.requires_audio # kwargs can override spec values return cls(name, spec.prompt, spec.choice_scores, client=client, **spec_kwargs, **kwargs) @@ -851,3 +864,69 @@ class Translation(SpecFileClassifier): """ pass + + +class SpeechClarity(SpecFileClassifier): + """Judge how clearly the agent speaks in a voice call recording, from the audio itself. + + Example: + ```python + import base64 + from openai import OpenAI + from autoevals import SpeechClarity + + with open("call.wav", "rb") as f: + audio = {"data": base64.b64encode(f.read()).decode(), "format": "wav"} + + result = SpeechClarity(client=OpenAI()).eval(output=None, audio=audio) + print(result.score) # 1 if clear, 0.5 if flaws are audible, 0 if words are lost + ``` + + Args: + audio: The call recording as OpenAI `input_audio`: base64 `data` and a `format` of "wav" or "mp3" + """ + + pass + + +class TurnTaking(SpecFileClassifier): + """Judge how well the agent takes turns in a voice call recording, from the audio itself. + + Example: + ```python + import base64 + from openai import OpenAI + from autoevals import TurnTaking + + with open("call.wav", "rb") as f: + audio = {"data": base64.b64encode(f.read()).decode(), "format": "wav"} + + result = TurnTaking(client=OpenAI()).eval(output=None, audio=audio) + print(result.score) # 1 if smooth, 0.5 if brief overlaps, 0 if caller words are lost + ``` + + Args: + audio: The call recording as OpenAI `input_audio`: base64 `data` and a `format` of "wav" or "mp3" + """ + + pass + + +class VoiceTaskSuccess(SpecFileClassifier): + """Judge whether a voice agent correctly completed the caller's request, from the trace's thread. + + The thread comes from `trace` and includes the agent's instructions, tool calls, and tool results. + + Example: + ```python + from autoevals import VoiceTaskSuccess + + async def voice_task_success(output, trace): + return await VoiceTaskSuccess().eval_async(output=output, trace=trace) + ``` + + Args: + trace: The voice call's trace + """ + + pass diff --git a/py/autoevals/test_llm.py b/py/autoevals/test_llm.py index 8e987efa..905b3f5c 100644 --- a/py/autoevals/test_llm.py +++ b/py/autoevals/test_llm.py @@ -9,7 +9,16 @@ from pydantic import BaseModel from autoevals import init -from autoevals.llm import Battle, Factuality, LLMClassifier, OpenAILLMClassifier, build_classification_tools +from autoevals.llm import ( + Battle, + Factuality, + LLMClassifier, + OpenAILLMClassifier, + SpeechClarity, + TurnTaking, + VoiceTaskSuccess, + build_classification_tools, +) from autoevals.oai import OpenAIV1Module, get_default_model from autoevals.thread_utils import compute_thread_template_vars, template_uses_thread_variables @@ -329,6 +338,65 @@ def test_factuality_client(): assert result.score == 1 +@respx.mock +def test_turn_taking_sends_audio(): + route = respx.route(method="POST", path__regex=r".*/chat/completions$").respond( + json={ + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 0, + "model": "gpt-audio", + "choices": [ + { + "index": 0, + "finish_reason": "tool_calls", + "message": { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_test", + "type": "function", + "function": { + "name": "select_choice", + "arguments": '{"reasons":"The agent talks over the caller.","choice":"C"}', + }, + } + ], + }, + } + ], + } + ) + + audio = {"data": "UklGRg==", "format": "wav"} + result = TurnTaking(client=OpenAI(api_key="test")).eval(output=None, audio=audio) + + assert result.score == 0 + body = json.loads(route.calls.last.request.content) + assert body["model"] == "gpt-audio" + assert body["messages"][-1] == {"role": "user", "content": [{"type": "input_audio", "input_audio": audio}]} + + +def test_voice_task_success_reads_thread_with_system(): + trace = _FakeTrace( + [ + {"role": "system", "content": "Only book tables for parties up to 8."}, + {"role": "user", "content": "Book a table for 12."}, + ] + ) + + request_args = VoiceTaskSuccess()._request_args(output=None, expected=None, trace=trace) + + assert "Only book tables for parties up to 8." in request_args["messages"][0]["content"] + assert len(request_args["messages"]) == 1 + + +def test_audio_scorers_require_audio(): + with pytest.raises(ValueError, match="SpeechClarity needs the call recording"): + SpeechClarity(client=OpenAI(api_key="test")).eval(output=None) + + @pytest.fixture(autouse=True) def reset_client(): yield diff --git a/templates/speech_clarity.yaml b/templates/speech_clarity.yaml new file mode 100644 index 00000000..40c66c02 --- /dev/null +++ b/templates/speech_clarity.yaml @@ -0,0 +1,14 @@ +prompt: |- + Listen to the attached recording of a conversation between an agent and a caller. Judge only the agent's speech. + + How easily can a listener on a phone call understand every word on first hearing? Consider garbled or cut-off words, distortion, dropouts, echo, and background noise. Do not fill in unclear words from context, and ignore accent unless it makes words hard to understand. + + (A) Words are clear and understandable without effort. + (B) Some flaws are audible, but the message remains easy to understand. + (C) Important words are unclear or lost. +choice_scores: + "A": 1.0 + "B": 0.5 + "C": 0.0 +model: gpt-audio +requires_audio: true diff --git a/templates/turn_taking.yaml b/templates/turn_taking.yaml new file mode 100644 index 00000000..44407525 --- /dev/null +++ b/templates/turn_taking.yaml @@ -0,0 +1,14 @@ +prompt: |- + Listen to the attached recording of a conversation between an agent and a caller. Judge only how the agent handles turn-taking. + + Consider whether the agent talks over the caller, cuts the caller off before they finish, or keeps talking when the caller interrupts. Brief backchannels such as "mm-hm" are fine unless they cover the caller's words. Ignore how long the agent takes to respond. + + (A) The agent takes turns smoothly: it lets the caller finish and stops promptly when interrupted. + (B) There are brief overlaps or a slow yield, but no caller words are lost. + (C) The agent talks over the caller, ignores an interruption, or cuts the caller off, so caller words are lost. +choice_scores: + "A": 1.0 + "B": 0.5 + "C": 0.0 +model: gpt-audio +requires_audio: true diff --git a/templates/voice_task_success.yaml b/templates/voice_task_success.yaml new file mode 100644 index 00000000..e44ab6c1 --- /dev/null +++ b/templates/voice_task_success.yaml @@ -0,0 +1,17 @@ +prompt: |- + You are judging a voice agent's conversation with a caller from its transcript. The transcript may include the agent's instructions, tool calls, and tool results. Here is the data: + [BEGIN DATA] + ************ + [Conversation]: {{thread_with_system}} + ************ + [END DATA] + + Judge whether the agent correctly understood and completed what the caller asked for. Check that it followed its instructions, used the right tools with the right arguments, and told the caller only what the tool results support. Saying an action succeeded is not proof that it did. If the request could not or should not be done, a clear refusal or handoff counts as completing it. Weigh correctness above conversational polish, and ignore speech quality and timing. + + (A) The agent completed the request correctly, followed its instructions, and everything it told the caller is supported. + (B) The agent completed the request, but with minor problems that did not change the outcome, such as unneeded steps or repeated questions. + (C) The agent did not complete the request, took a wrong or unauthorized action, broke its instructions, or told the caller something false or unsupported. +choice_scores: + "A": 1.0 + "B": 0.5 + "C": 0.0