Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions haystack/components/evaluators/answer_exact_match.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,10 @@ def run(self, ground_truth_answers: list[str], predicted_answers: list[str]) ->
- `score` - A number from 0.0 to 1.0 that represents the proportion of questions where any predicted
answer matched one of the ground truth answers.
"""
if len(ground_truth_answers) == 0 or len(predicted_answers) == 0:
msg = "ground_truth_answers and predicted_answers must be provided."
raise ValueError(msg)

if not len(ground_truth_answers) == len(predicted_answers):
raise ValueError("The length of ground_truth_answers and predicted_answers must be the same.")

Expand Down
4 changes: 4 additions & 0 deletions haystack/components/evaluators/document_map.py
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,10 @@ def run(
- `individual_scores` - A list of numbers from 0.0 to 1.0 that represents how high retrieved documents
are ranked.
"""
if len(ground_truth_documents) == 0 or len(retrieved_documents) == 0:
msg = "ground_truth_documents and retrieved_documents must be provided."
raise ValueError(msg)

if len(ground_truth_documents) != len(retrieved_documents):
msg = "The length of ground_truth_documents and retrieved_documents must be the same."
raise ValueError(msg)
Expand Down
4 changes: 4 additions & 0 deletions haystack/components/evaluators/document_mrr.py
Original file line number Diff line number Diff line change
Expand Up @@ -106,6 +106,10 @@ def run(
- `individual_scores` - A list of numbers from 0.0 to 1.0 that represents how high the first retrieved
document is ranked.
"""
if len(ground_truth_documents) == 0 or len(retrieved_documents) == 0:
msg = "ground_truth_documents and retrieved_documents must be provided."
raise ValueError(msg)

if len(ground_truth_documents) != len(retrieved_documents):
msg = "The length of ground_truth_documents and retrieved_documents must be the same."
raise ValueError(msg)
Expand Down
4 changes: 4 additions & 0 deletions haystack/components/evaluators/document_recall.py
Original file line number Diff line number Diff line change
Expand Up @@ -172,6 +172,10 @@ def run(
- `individual_scores` - A list of numbers from 0.0 to 1.0 that represents the proportion of matching
documents retrieved. If the mode is `single_hit`, the individual scores are 0 or 1.
"""
if len(ground_truth_documents) == 0 or len(retrieved_documents) == 0:
msg = "ground_truth_documents and retrieved_documents must be provided."
raise ValueError(msg)

if len(ground_truth_documents) != len(retrieved_documents):
msg = "The length of ground_truth_documents and retrieved_documents must be the same."
raise ValueError(msg)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
---
fixes:
- |
``DocumentMAPEvaluator``, ``DocumentMRREvaluator``, ``DocumentRecallEvaluator`` and
``AnswerExactMatchEvaluator`` now raise ``ValueError`` when given no questions, instead of
``ZeroDivisionError``. This matches ``DocumentNDCGEvaluator``. A question whose own document
or answer list is empty is unchanged.
49 changes: 49 additions & 0 deletions test/components/evaluators/test_empty_input.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,49 @@
# SPDX-FileCopyrightText: 2022-present deepset GmbH <info@deepset.ai>
#
# SPDX-License-Identifier: Apache-2.0

import pytest

from haystack import Document
from haystack.components.evaluators.answer_exact_match import AnswerExactMatchEvaluator
from haystack.components.evaluators.document_map import DocumentMAPEvaluator
from haystack.components.evaluators.document_mrr import DocumentMRREvaluator
from haystack.components.evaluators.document_recall import DocumentRecallEvaluator, RecallMode


@pytest.mark.parametrize(
"evaluator",
[
DocumentMAPEvaluator(),
DocumentMRREvaluator(),
DocumentRecallEvaluator(mode=RecallMode.SINGLE_HIT),
DocumentRecallEvaluator(mode=RecallMode.MULTI_HIT),
],
)
def test_document_evaluators_reject_top_level_empty_input(evaluator):
with pytest.raises(ValueError, match="must be provided"):
evaluator.run(ground_truth_documents=[], retrieved_documents=[])

with pytest.raises(ValueError, match="must be provided"):
evaluator.run(ground_truth_documents=[], retrieved_documents=[[Document(content="x")]])

with pytest.raises(ValueError, match="must be provided"):
evaluator.run(ground_truth_documents=[[Document(content="x")]], retrieved_documents=[])


def test_document_evaluators_still_score_a_question_with_no_documents():
ground_truth = [[]]
retrieved = [[Document(content="x")]]
for evaluator in (DocumentMAPEvaluator(), DocumentMRREvaluator(), DocumentRecallEvaluator()):
result = evaluator.run(ground_truth_documents=ground_truth, retrieved_documents=retrieved)
assert result["score"] == 0.0
assert result["individual_scores"] == [0.0]


def test_answer_exact_match_rejects_top_level_empty_input():
evaluator = AnswerExactMatchEvaluator()
with pytest.raises(ValueError, match="must be provided"):
evaluator.run(ground_truth_answers=[], predicted_answers=[])

result = evaluator.run(ground_truth_answers=["Berlin"], predicted_answers=["Berlin"])
assert result == {"individual_scores": [1], "score": 1.0}