diff --git a/AGENTS.md b/AGENTS.md index 85089885..1d1c3a7f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -54,7 +54,7 @@ The library maintains parallel implementations in TypeScript (`js/`) and Python ### Key Modules (both languages) -- `llm.ts` / `llm.py` - LLM-as-a-judge scorers (Factuality, Battle, ClosedQA, Humor, Security, Sql, Summary, Translation) +- `llm.ts` / `llm.py` - LLM-as-a-judge scorers (Factuality, Battle, ClosedQA, Humor, Security, Sql, Summary, Translation, VoiceTaskSuccess) - `ragas.ts` / `ragas.py` - RAG evaluation metrics (ContextRelevancy, Faithfulness, AnswerRelevancy, etc.) - `string.ts` / `string.py` - Text similarity (Levenshtein, EmbeddingSimilarity) - `json.ts` / `json.py` - JSON validation and diff diff --git a/README.md b/README.md index 7b2a179c..bbc25677 100644 --- a/README.md +++ b/README.md @@ -332,6 +332,7 @@ Eval( - Summarization - SQL - Translation +- Voice task success - Fine-tuned binary classifiers ### RAG evaluations diff --git a/SCORERS.md b/SCORERS.md index 56893245..e846ab44 100644 --- a/SCORERS.md +++ b/SCORERS.md @@ -192,6 +192,24 @@ Evaluates translation quality. - `1.0` = Excellent translation - `0.0` = Poor translation +### VoiceTaskSuccess + +Evaluates whether a voice agent correctly completed the caller's request, from the whole call. + +**Parameters:** + +- `thread_with_system` (array, optional): The call's messages +- `trace` (Trace, optional): The voice call's trace. Used when `thread_with_system` is missing; the conversation comes from `trace.getThread()` / `trace.get_thread()` +- `model` (string, optional): Model to use + +The judge sees the messages as JSON, including tool calls and results. With no conversation, the score is `null`. + +**Score Range:** 0-1 + +- `1.0` = Completed correctly +- `0.5` = Completed with minor problems +- `0.0` = Not completed, or wrong or unsupported + --- ## RAG (Retrieval-Augmented Generation) scorers diff --git a/js/llm.ts b/js/llm.ts index 1e572324..2ad6978d 100644 --- a/js/llm.ts +++ b/js/llm.ts @@ -492,3 +492,32 @@ export const Translation = buildLLMClassifier<{ language: string; input: string; }>("Translation", "translation"); + +type VoiceTaskSuccessArgs = { + thread_with_system?: unknown[]; +}; + +const voiceTaskSuccessClassifier = LLMClassifierFromSpecFile<{ + thread_with_system: string; +}>("VoiceTaskSuccess", "voice_task_success"); + +/** + * Test whether a voice agent correctly completed the caller's request, from the call in + * `thread_with_system` or, if that's missing, the conversation in the `trace`. + * Returns a null score when there is no conversation. + */ +export const VoiceTaskSuccess = makePartial< + string, + LLMClassifierArgs +>(async ({ thread_with_system, trace, ...args }) => { + const messages = thread_with_system ?? (await trace?.getThread()); + + if (!messages?.length) { + return { name: "VoiceTaskSuccess", score: null }; + } + + return voiceTaskSuccessClassifier({ + ...args, + thread_with_system: JSON.stringify(messages), + }); +}, "VoiceTaskSuccess"); diff --git a/js/manifest.ts b/js/manifest.ts index c21d3703..cc3258c1 100644 --- a/js/manifest.ts +++ b/js/manifest.ts @@ -9,6 +9,7 @@ import { Sql, Summary, Translation, + VoiceTaskSuccess, } from "./llm"; import { NumericDiff } from "./number"; import { EmbeddingSimilarity, Levenshtein } from "./string"; @@ -101,6 +102,12 @@ export const Evaluators: { template: templates.translation, requiresExtraParams: true, }, + { + method: VoiceTaskSuccess, + description: + "Test whether a voice agent correctly completed the caller's request, from the whole call. Run it with trace scope.", + template: templates.voice_task_success, + }, ], }, { diff --git a/js/templates.ts b/js/templates.ts index 69f4637c..83f1281b 100644 --- a/js/templates.ts +++ b/js/templates.ts @@ -10,6 +10,7 @@ import security from "../templates/security.yaml"; import sql from "../templates/sql.yaml"; import summary from "../templates/summary.yaml"; import translation from "../templates/translation.yaml"; +import voice_task_success from "../templates/voice_task_success.yaml"; export const modelGradedSpecSchema = z.object({ prompt: z.string(), @@ -32,6 +33,7 @@ const templateStrings = { sql, summary, translation, + voice_task_success, } as const; // eslint-disable-next-line @typescript-eslint/consistent-type-assertions diff --git a/py/autoevals/llm.py b/py/autoevals/llm.py index 1a38712a..cda70ba0 100644 --- a/py/autoevals/llm.py +++ b/py/autoevals/llm.py @@ -851,3 +851,52 @@ class Translation(SpecFileClassifier): """ pass + + +class VoiceTaskSuccess(ScorerWithPartial): + """Rate whether a voice agent correctly completed the caller's request, from the whole call. + + The call comes from `thread_with_system` when given, else `trace.get_thread()`. + The score is None when there is no conversation. + + Example: + ```python + from autoevals import VoiceTaskSuccess + + result = await VoiceTaskSuccess().eval_async(output=None, trace=trace) + print(result.score) # 1 if done correctly, 0.5 if done with minor problems, 0 if not done or wrong + ``` + + Args: + thread_with_system: Optional list of the call's messages + trace: Trace of the voice conversation, used when `thread_with_system` is missing + """ + + def __init__(self, **kwargs): + template_path = os.path.join(SCRIPT_DIR, "templates", "voice_task_success.yaml") + self._classifier = LLMClassifier.from_spec_file("VoiceTaskSuccess", template_path, **kwargs) + + def _run_eval_sync(self, output, expected=None, thread_with_system=None, trace=None, **kwargs): + messages = thread_with_system + if messages is None and trace is not None: + try: + asyncio.get_running_loop() + except RuntimeError: + messages = list(asyncio.run(trace.get_thread())) + else: + raise RuntimeError("trace.get_thread() is async; use eval_async() when already inside an event loop") + + if not messages: + return Score(name="VoiceTaskSuccess", score=None) + conversation = json.dumps(messages, separators=(",", ":"), ensure_ascii=False) + return self._classifier.eval(output, expected, thread_with_system=conversation, **kwargs) + + async def _run_eval_async(self, output, expected=None, thread_with_system=None, trace=None, **kwargs): + messages = thread_with_system + if messages is None and trace is not None: + messages = list(await trace.get_thread()) + + if not messages: + return Score(name="VoiceTaskSuccess", score=None) + conversation = json.dumps(messages, separators=(",", ":"), ensure_ascii=False) + return await self._classifier.eval_async(output, expected, thread_with_system=conversation, **kwargs) diff --git a/templates/voice_task_success.yaml b/templates/voice_task_success.yaml new file mode 100644 index 00000000..d6190549 --- /dev/null +++ b/templates/voice_task_success.yaml @@ -0,0 +1,17 @@ +prompt: |- + You are judging a voice agent's conversation with a caller from its transcript. The transcript may include the agent's instructions, tool calls, and tool results. Here is the data: + [BEGIN DATA] + ************ + [Conversation]: + {{thread_with_system}} + ************ + [END DATA] + + Judge whether the agent correctly understood and completed what the caller asked for. Check that it followed its instructions, used the right tools with the right arguments, and told the caller only what the tool results support. Saying an action succeeded is not proof that it did. If the request could not or should not be done, a clear refusal or handoff counts as completing it. Weigh correctness above conversational polish, and ignore speech quality and timing. + (A) The agent completed the request correctly, followed its instructions, and everything it told the caller is supported. + (B) The agent completed the request, but with minor problems that did not change the outcome, such as unneeded steps or repeated questions. + (C) The agent did not complete the request, took a wrong or unauthorized action, broke its instructions, or told the caller something false or unsupported. +choice_scores: + "A": 1.0 + "B": 0.5 + "C": 0.0