Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions py/autoevals/ragas.py
Original file line number Diff line number Diff line change
Expand Up @@ -995,7 +995,7 @@ async def _run_eval_async(self, output, expected=None, input=None, context=None,

return Score(
name=self._name(),
score=sum([s["verdict"] for s in faithfulness]) / len(faithfulness),
score=(sum([s["verdict"] for s in faithfulness]) / len(faithfulness)) if faithfulness else 0,
metadata={
"statements": statements,
"faithfulness": faithfulness,
Expand All @@ -1017,7 +1017,7 @@ def _run_eval_sync(self, output, expected=None, input=None, context=None, **kwar

return Score(
name=self._name(),
score=sum([s["verdict"] for s in faithfulness]) / len(faithfulness),
score=(sum([s["verdict"] for s in faithfulness]) / len(faithfulness)) if faithfulness else 0,
metadata={
"statements": statements,
"faithfulness": faithfulness,
Expand Down
39 changes: 39 additions & 0 deletions py/autoevals/test_ragas.py
Original file line number Diff line number Diff line change
Expand Up @@ -171,6 +171,45 @@ def fake_extract_faithfulness(context, statements, client=None, **extra_args):
assert captured_answer == "Paris is the capital of France."


@pytest.mark.parametrize("is_async", [False, True])
def test_faithfulness_scores_zero_when_judge_finds_no_statements(monkeypatch, is_async):
"""A judge that finds no statements scores 0 instead of dividing by zero.

The TypeScript scorer already handles this: ``faithfulness.length ? ... : 0``
in ``js/ragas.ts``. The Python scorer divides by ``len(faithfulness)``
unguarded, so an empty verdict list aborts the whole eval run with a
ZeroDivisionError instead of reporting an unfaithful answer.
"""

def fake_extract_statements(*args, **kwargs):
return {"statements": []}

def fake_extract_faithfulness(*args, **kwargs):
return {"faithfulness": []}

async def fake_aextract_statements(*args, **kwargs):
return {"statements": []}

async def fake_aextract_faithfulness(*args, **kwargs):
return {"faithfulness": []}

monkeypatch.setattr(ragas_module, "extract_statements", fake_extract_statements)
monkeypatch.setattr(ragas_module, "extract_faithfulness", fake_extract_faithfulness)
monkeypatch.setattr(ragas_module, "aextract_statements", fake_aextract_statements)
monkeypatch.setattr(ragas_module, "aextract_faithfulness", fake_aextract_faithfulness)

scorer = Faithfulness()
args = {
"input": "What is the capital of France?",
"output": "",
"context": "Paris is the capital of France.",
}
score = asyncio.run(scorer.eval_async(**args)) if is_async else scorer.eval(**args)

assert score.score == 0
assert score.metadata["faithfulness"] == []


@respx.mock
def test_answer_correctness_uses_custom_embedding_model():
"""Test that AnswerCorrectness passes embedding_model parameter through to embeddings API."""
Expand Down