diff --git a/AGENTS.md b/AGENTS.md index 36ad4b9..0302db2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -54,7 +54,7 @@ The library maintains parallel implementations in TypeScript (`js/`) and Python ### Key Modules (both languages) -- `llm.ts` / `llm.py` - LLM-as-a-judge scorers (Factuality, Battle, ClosedQA, Humor, Security, Sql, Summary, Translation, VoiceTaskSuccess) +- `llm.ts` / `llm.py` - LLM-as-a-judge scorers (Factuality, Battle, ClosedQA, Humor, Security, Sql, Summary, Translation, VoiceTaskSuccess, SpeechClarity) - `ragas.ts` / `ragas.py` - RAG evaluation metrics (ContextRelevancy, Faithfulness, AnswerRelevancy, etc.) - `string.ts` / `string.py` - Text similarity (Levenshtein, EmbeddingSimilarity) - `json.ts` / `json.py` - JSON validation and diff @@ -69,6 +69,7 @@ YAML templates in `templates/` define LLM classifier prompts. Templates use Must - Prompt rendering with chain-of-thought (CoT) suffix - Tool-based response parsing via `select_choice` function - Score mapping from choice letters to numeric scores +- An optional `audio:` path (e.g. `input.audio`) to `{data, content_type}` bytes, sent as a file part; missing audio gives a `null` score ### Python Scorer Pattern diff --git a/README.md b/README.md index 51a4eea..ae0cdc0 100644 --- a/README.md +++ b/README.md @@ -333,6 +333,7 @@ Eval( - SQL - Translation - Voice task success +- Speech clarity - Fine-tuned binary classifiers ### RAG evaluations diff --git a/SCORERS.md b/SCORERS.md index a4e7d26..1353235 100644 --- a/SCORERS.md +++ b/SCORERS.md @@ -211,6 +211,35 @@ The judge sees the messages as JSON, including tool calls and results. With no c - `0.5` = Completed with minor problems - `0.0` = Not completed, or wrong or unsupported +### SpeechClarity + +Evaluates how clearly the agent speaks, from a recording of the whole call. + +**Parameters:** + +- `input.audio` (object or list): The recording, as `{ data, content_type }`, where `data` is the bytes (`Uint8Array` in TypeScript, `bytes` in Python) and `content_type` is an `audio/*` type such as `audio/ogg`. A list of these, such as a recording split into chunks, is sent as separate files in list order. +- `model` (string, optional): Model to use (default: `gemini-3.8-flash`). It must accept audio as a Chat Completions `file` part. + +Without `input.audio`, or with an empty list, the score is `null`. + +**Score Range:** 0-1 + +- `1.0` = Clear and understandable without effort +- `0.5` = Audible flaws, but easy to understand +- `0.0` = Important words unclear or lost + +**Example:** + +```python +from pathlib import Path +from autoevals import SpeechClarity + +audio = {"data": Path("call.ogg").read_bytes(), "content_type": "audio/ogg"} +result = SpeechClarity().eval(input={"audio": audio}, output=None) +``` + +To build your own audio judge, set `audio` to the recording's path in the scorer's arguments: `audio: input.audio` in a template, `audio="input.audio"` for `LLMClassifier`, or `audio: "input.audio"` for `LLMClassifierFromTemplate`. + --- ## RAG (Retrieval-Augmented Generation) scorers diff --git a/js/llm.test.ts b/js/llm.test.ts index 446afc0..1f32ae2 100644 --- a/js/llm.test.ts +++ b/js/llm.test.ts @@ -9,6 +9,7 @@ import { DEFAULT_MODEL, LLMClassifierFromTemplate, OpenAIClassifier, + SpeechClarity, templateUsesThreadVariables, VoiceTaskSuccess, } from "../js/llm"; @@ -724,3 +725,63 @@ describe("VoiceTaskSuccess", () => { expect(score.score).toBeNull(); }); }); + +describe("SpeechClarity", () => { + test("sends each audio as a file to gemini-3.8-flash through chat completions", async () => { + let body: any; + server.use( + http.post( + "https://api.openai.com/v1/chat/completions", + async ({ request }) => { + body = await request.json(); + return HttpResponse.json({ + choices: [ + { + message: { + role: "assistant", + tool_calls: [ + { + id: "call_test", + type: "function", + function: { + name: "select_choice", + arguments: '{"reasons":"Clear.","choice":"A"}', + }, + }, + ], + }, + }, + ], + }); + }, + ), + ); + + const score = await SpeechClarity({ + input: { + audio: [ + { data: new Uint8Array([1, 2, 3]), content_type: "audio/ogg" }, + { data: new Uint8Array([4, 5, 6]), content_type: "audio/ogg" }, + ], + }, + output: undefined, + openAiApiKey: "test-api-key", + }); + + expect(score.score).toBe(1); + expect(body.model).toBe("gemini-3.8-flash"); + expect(body.messages[0].content[0].type).toBe("text"); + expect( + body.messages[0].content.slice(1).map((p: any) => p.file.file_data), + ).toEqual(["data:audio/ogg;base64,AQID", "data:audio/ogg;base64,BAUG"]); + }); + + test("skips when there is no audio", async () => { + const score = await SpeechClarity({ + input: { text: "hello" }, + output: undefined, + openAiApiKey: "test-api-key", + }); + expect(score.score).toBeNull(); + }); +}); diff --git a/js/llm.ts b/js/llm.ts index 1eff593..9978eeb 100644 --- a/js/llm.ts +++ b/js/llm.ts @@ -7,6 +7,7 @@ import { } from "./oai"; import { ModelGradedSpec, templates } from "./templates"; import { + ChatCompletionContentPart, ChatCompletionMessage, ChatCompletionMessageParam, ChatCompletionTool, @@ -62,6 +63,42 @@ function filterSystemMessagesFromThread(thread: unknown[]): unknown[] { }); } +function getPath(args: unknown, path: string): unknown { + let value = args; + for (const key of path.split(".")) { + value = + value && typeof value === "object" && !Array.isArray(value) + ? Reflect.get(value, key) + : undefined; + } + return value; +} + +export type Audio = { data: Uint8Array; content_type: string }; + +function audioPart(audio: unknown): ChatCompletionContentPart.File { + const data = Reflect.get(Object(audio), "data"); + if (!(data instanceof Uint8Array)) { + throw new TypeError( + "Audio must be an object with `data` bytes and a `content_type`", + ); + } + const contentType = String(Reflect.get(Object(audio), "content_type") ?? "") + .split(";")[0] + .trim() + .toLowerCase(); + if (!contentType.startsWith("audio/")) { + throw new Error( + `Audio must have an audio/* content type, got "${contentType}"`, + ); + } + const base64 = Buffer.from(data).toString("base64"); + return { + type: "file", + file: { file_data: `data:${contentType};base64,${base64}` }, + }; +} + const NO_COT_SUFFIX = "Answer the question by calling `select_choice` with a single choice from {{__choices}}."; @@ -142,6 +179,7 @@ export type OpenAIClassifierArgs = { messages: ChatCompletionMessageParam[]; choiceScores: Record; classificationTools: ChatCompletionTool[]; + audioFiles?: ChatCompletionContentPart.File[]; cache?: ChatCache; } & LLMArgs & RenderArgs; @@ -174,6 +212,7 @@ export async function OpenAIClassifier( reasoningEnabled, reasoningBudget, useResponsesApi, + audioFiles, cache, ...remainingRenderArgs } = remaining; @@ -212,6 +251,13 @@ export async function OpenAIClassifier( }; const messages = renderMessages(messagesArg, renderArgs); + if (audioFiles) { + const last = messages[messages.length - 1]; + messages[messages.length - 1] = { + role: "user", + content: [{ type: "text", text: String(last.content) }, ...audioFiles], + }; + } const resp = await cachedChatCompletion( { @@ -306,6 +352,7 @@ export function LLMClassifierFromTemplate({ reasoningEnabled, reasoningBudget, useResponsesApi, + audio, }: { name: string; promptTemplate: string; @@ -318,11 +365,23 @@ export function LLMClassifierFromTemplate({ reasoningEnabled?: boolean; reasoningBudget?: number; useResponsesApi?: boolean; + audio?: string; }): Scorer> { const choiceStrings = Object.keys(choiceScores); const ret = async ( runtimeArgs: ScorerArgs>, ) => { + const audioValue = audio ? getPath(runtimeArgs, audio) : undefined; + const audioList = + audioValue == null + ? [] + : Array.isArray(audioValue) + ? audioValue + : [audioValue]; + if (audio && audioList.length === 0) { + return { name, score: null }; + } + const useCoT = runtimeArgs.useCoT ?? useCoTArg ?? true; // Use runtime model > template model > configured default model const model = runtimeArgs.model ?? modelArg ?? getDefaultModel(); @@ -374,6 +433,7 @@ export function LLMClassifierFromTemplate({ // Since the logic is a bit funky for computing this, include // it at the end to prevent overrides useCoT, + audioFiles: audio ? audioList.map(audioPart) : undefined, }; return await OpenAIClassifier(classifierArgs); @@ -398,6 +458,7 @@ export function LLMClassifierFromSpec( useCoT: spec.use_cot, temperature: spec.temperature, maxTokens: spec.max_tokens, + audio: spec.audio, }); } @@ -409,10 +470,10 @@ export function LLMClassifierFromSpecFile( return LLMClassifierFromSpec(name, doc); } -function buildLLMClassifier( +function buildLLMClassifier( name: string, templateName: keyof typeof templates, -): ScorerWithPartial> { +): ScorerWithPartial> { if (!(templateName in templates)) { throw new Error(`Model template ${name} not found`); } @@ -521,3 +582,11 @@ export const VoiceTaskSuccess = makePartial< thread_with_system: JSON.stringify(messages), }); }, "VoiceTaskSuccess"); + +/** + * Test how clearly an agent speaks, from a recording of the whole conversation (`input.audio`), given as one audio or a list of chunks in recording order. + */ +export const SpeechClarity = buildLLMClassifier< + { input: { audio?: Audio | Audio[]; [key: string]: unknown } }, + unknown +>("SpeechClarity", "speech_clarity"); diff --git a/js/manifest.ts b/js/manifest.ts index 29c1cee..25b8d54 100644 --- a/js/manifest.ts +++ b/js/manifest.ts @@ -6,6 +6,7 @@ import { Humor, Possible, Security, + SpeechClarity, Sql, Summary, Translation, @@ -84,6 +85,12 @@ export const Evaluators: { description: "Test whether an output is malicious.", template: templates.security, }, + { + method: SpeechClarity, + description: + "Test how clearly an agent's speech can be understood, from a recording of the whole conversation (`input.audio`).", + template: templates.speech_clarity, + }, { method: Sql, description: diff --git a/js/templates.ts b/js/templates.ts index 83f1281..9d78b96 100644 --- a/js/templates.ts +++ b/js/templates.ts @@ -7,6 +7,7 @@ import factuality from "../templates/factuality.yaml"; import humor from "../templates/humor.yaml"; import possible from "../templates/possible.yaml"; import security from "../templates/security.yaml"; +import speech_clarity from "../templates/speech_clarity.yaml"; import sql from "../templates/sql.yaml"; import summary from "../templates/summary.yaml"; import translation from "../templates/translation.yaml"; @@ -19,6 +20,7 @@ export const modelGradedSpecSchema = z.object({ use_cot: z.boolean().optional(), temperature: z.number().optional(), max_tokens: z.number().optional(), + audio: z.string().optional(), }); export type ModelGradedSpec = z.infer; @@ -30,6 +32,7 @@ const templateStrings = { humor, possible, security, + speech_clarity, sql, summary, translation, diff --git a/py/autoevals/llm.py b/py/autoevals/llm.py index cda70ba..4bda2e1 100644 --- a/py/autoevals/llm.py +++ b/py/autoevals/llm.py @@ -46,6 +46,7 @@ """ import asyncio +import base64 import inspect import json import os @@ -133,6 +134,30 @@ def build_classification_tools(useCoT, choice_strings): ] +def _get_path(args, path): + value = args + for key in path.split("."): + value = value.get(key) if isinstance(value, dict) else None + return value + + +def _get_audio(args, path): + audio = _get_path(args, path) + if audio is None: + return [] + return audio if isinstance(audio, list) else [audio] + + +def _audio_part(audio): + if not isinstance(audio, dict) or not isinstance(audio.get("data"), (bytes, bytearray)): + raise TypeError("Audio must be a dict with `data` bytes and a `content_type`") + content_type = str(audio.get("content_type", "")).split(";")[0].strip().lower() + if not content_type.startswith("audio/"): + raise ValueError(f"Audio must have an audio/* content type, got {content_type!r}") + data = base64.b64encode(audio["data"]).decode() + return {"type": "file", "file": {"file_data": f"data:{content_type};base64,{data}"}} + + class OpenAIScorer(ScorerWithPartial): def __init__( self, @@ -181,6 +206,7 @@ def __init__( reasoning_enabled=None, reasoning_budget=None, use_responses_api=None, + audio=None, engine=None, api_key=None, base_url=None, @@ -198,6 +224,7 @@ def __init__( self.model = model self.engine = engine self.messages = messages + self.audio = audio if max_tokens is not None: self.extra_args["max_tokens"] = max(max_tokens, 5) @@ -234,13 +261,22 @@ def _build_args(self, output, expected, **kwargs): def _render_messages(self, **kwargs): kwargs.update(self.render_args) - return [ + messages = [ { **m, "content": chevron.render(m["content"].strip(), kwargs, warn=True), } for m in self.messages ] + if self.audio: + text = messages[-1]["content"] + files = [_audio_part(a) for a in _get_audio(kwargs, self.audio)] + messages[-1]["content"] = [{"type": "text", "text": text}, *files] + return messages + + def _missing_audio(self, output, expected, **kwargs): + args = {"output": output, "expected": expected, **kwargs, **self.render_args} + return self.audio and not _get_audio(args, self.audio) def _request_args(self, output, expected, **kwargs): ret = { @@ -284,11 +320,15 @@ def _postprocess_response(self, resp): raise ValueError("Empty response from OpenAI") async def _run_eval_async(self, output, expected, **kwargs): + if self._missing_audio(output, expected, **kwargs): + return Score(name=self.name, score=None) return self._postprocess_response( await arun_cached_request(**(await self._request_args_async(output, expected, **kwargs))) ) def _run_eval_sync(self, output, expected, **kwargs): + if self._missing_audio(output, expected, **kwargs): + return Score(name=self.name, score=None) return self._postprocess_response(run_cached_request(**self._request_args(output, expected, **kwargs))) @@ -301,6 +341,7 @@ class ModelGradedSpec: use_cot: bool | None = None temperature: float | None = None max_tokens: int | None = None + audio: str | None = None class LLMClassifier(OpenAILLMClassifier): @@ -344,6 +385,7 @@ class LLMClassifier(OpenAILLMClassifier): reasoning_effort: Controls reasoning depth for o-series models (e.g., "low", "medium", "high"). reasoning_enabled: Enable extended thinking for supported models (e.g., Claude). Defaults to None. reasoning_budget: Token allocation for model's internal reasoning. Defaults to None. + audio: Path in the arguments (e.g. `input.audio`) to a dict with `data` bytes and an audio `content_type`, such as `audio/ogg`, or a list of these dicts, each sent as its own file in order. Missing audio or an empty list skips the score. engine: Deprecated by OpenAI. Use model instead. api_key: Deprecated. Use client instead. base_url: Deprecated. Use client instead. @@ -371,6 +413,7 @@ def __init__( reasoning_enabled=None, reasoning_budget=None, use_responses_api=None, + audio=None, engine=None, api_key=None, base_url=None, @@ -403,6 +446,7 @@ def __init__( reasoning_enabled=reasoning_enabled, reasoning_budget=reasoning_budget, use_responses_api=use_responses_api, + audio=audio, engine=engine, api_key=api_key, base_url=base_url, @@ -484,8 +528,11 @@ def from_spec(cls, name: str, spec: ModelGradedSpec, client: Client | None = Non spec_kwargs["temperature"] = spec.temperature if spec.max_tokens is not None: spec_kwargs["max_tokens"] = spec.max_tokens + if spec.audio is not None: + spec_kwargs["audio"] = spec.audio # kwargs can override spec values - return cls(name, spec.prompt, spec.choice_scores, client=client, **spec_kwargs, **kwargs) + spec_kwargs.update(kwargs) + return cls(name, spec.prompt, spec.choice_scores, client=client, **spec_kwargs) @classmethod def from_spec_file(cls, name: str, path: str, client: Client | None = None, **kwargs): @@ -900,3 +947,23 @@ async def _run_eval_async(self, output, expected=None, thread_with_system=None, return Score(name="VoiceTaskSuccess", score=None) conversation = json.dumps(messages, separators=(",", ":"), ensure_ascii=False) return await self._classifier.eval_async(output, expected, thread_with_system=conversation, **kwargs) + + +class SpeechClarity(SpecFileClassifier): + """Rate how clearly an agent speaks in a recording of the whole conversation. + + Example: + ```python + from pathlib import Path + from autoevals import SpeechClarity + + audio = {"data": Path("call.ogg").read_bytes(), "content_type": "audio/ogg"} + result = SpeechClarity().eval(input={"audio": audio}, output=None) + print(result.score) # 1 if clear, 0.5 if flawed but easy to follow, 0 if words are lost + ``` + + Args: + input: Dict whose `audio` is the recording, as `{"data": bytes, "content_type": "audio/ogg"}`, or a list of them in recording order + """ + + pass diff --git a/py/autoevals/test_llm.py b/py/autoevals/test_llm.py index d06366a..f7eb746 100644 --- a/py/autoevals/test_llm.py +++ b/py/autoevals/test_llm.py @@ -15,6 +15,7 @@ Factuality, LLMClassifier, OpenAILLMClassifier, + SpeechClarity, VoiceTaskSuccess, build_classification_tools, ) @@ -818,3 +819,54 @@ def test_voice_task_success_sends_the_traces_conversation_as_json(): def test_voice_task_success_skips_without_conversation(): assert VoiceTaskSuccess().eval(output=None, trace=_FakeTrace([])).score is None + + +@respx.mock +def test_speech_clarity_sends_each_audio_as_a_file_to_chat_completions(): + route = respx.route(method="POST", path__regex=r".*/chat/completions$").respond( + json={ + "choices": [ + { + "message": { + "role": "assistant", + "tool_calls": [ + { + "id": "call_test", + "type": "function", + "function": { + "name": "select_choice", + "arguments": '{"reasons":"Clear.","choice":"A"}', + }, + } + ], + }, + } + ], + } + ) + result = SpeechClarity(client=OpenAI(api_key="test", base_url="https://api.openai.com/v1")).eval( + input={ + "audio": [ + {"data": b"\x01\x02\x03", "content_type": "audio/ogg"}, + {"data": b"\x04\x05\x06", "content_type": "audio/ogg"}, + ] + }, + output=None, + ) + + assert result.score == 1 + body = json.loads(route.calls.last.request.content) + assert body["model"] == "gemini-3.8-flash" + assert body["messages"][0]["content"][0]["type"] == "text" + assert [p["file"]["file_data"] for p in body["messages"][0]["content"][1:]] == [ + "data:audio/ogg;base64,AQID", + "data:audio/ogg;base64,BAUG", + ] + + +def test_speech_clarity_accepts_model_override(): + assert SpeechClarity(model="gemini-2.5-pro").model == "gemini-2.5-pro" + + +def test_speech_clarity_skips_without_audio(): + assert SpeechClarity().eval(input={"text": "hello"}, output=None).score is None diff --git a/templates/speech_clarity.yaml b/templates/speech_clarity.yaml new file mode 100644 index 0000000..9f9e23f --- /dev/null +++ b/templates/speech_clarity.yaml @@ -0,0 +1,12 @@ +prompt: |- + Listen to the attached recording of a conversation between an agent and a caller. Judge only the agent's speech. + How easily can a listener on a phone call understand every word on first hearing? Consider garbled or cut-off words, distortion, dropouts, echo, and background noise. Do not fill in unclear words from context, and ignore accent unless it makes words hard to understand. + (A) Words are clear and understandable without effort. + (B) Some flaws are audible, but the message remains easy to understand. + (C) Important words are unclear or lost. +choice_scores: + "A": 1.0 + "B": 0.5 + "C": 0.0 +model: gemini-3.8-flash +audio: input.audio