diff --git a/haystack/components/evaluators/answer_exact_match.py b/haystack/components/evaluators/answer_exact_match.py index a2bc57e549b..b993dd38f50 100644 --- a/haystack/components/evaluators/answer_exact_match.py +++ b/haystack/components/evaluators/answer_exact_match.py @@ -53,6 +53,10 @@ def run(self, ground_truth_answers: list[str], predicted_answers: list[str]) -> - `score` - A number from 0.0 to 1.0 that represents the proportion of questions where any predicted answer matched one of the ground truth answers. """ + if len(ground_truth_answers) == 0 or len(predicted_answers) == 0: + msg = "ground_truth_answers and predicted_answers must be provided." + raise ValueError(msg) + if not len(ground_truth_answers) == len(predicted_answers): raise ValueError("The length of ground_truth_answers and predicted_answers must be the same.") diff --git a/haystack/components/evaluators/document_map.py b/haystack/components/evaluators/document_map.py index eb5180fe909..8963df7294d 100644 --- a/haystack/components/evaluators/document_map.py +++ b/haystack/components/evaluators/document_map.py @@ -108,6 +108,10 @@ def run( - `individual_scores` - A list of numbers from 0.0 to 1.0 that represents how high retrieved documents are ranked. """ + if len(ground_truth_documents) == 0 or len(retrieved_documents) == 0: + msg = "ground_truth_documents and retrieved_documents must be provided." + raise ValueError(msg) + if len(ground_truth_documents) != len(retrieved_documents): msg = "The length of ground_truth_documents and retrieved_documents must be the same." raise ValueError(msg) diff --git a/haystack/components/evaluators/document_mrr.py b/haystack/components/evaluators/document_mrr.py index db51583521a..3550768115e 100644 --- a/haystack/components/evaluators/document_mrr.py +++ b/haystack/components/evaluators/document_mrr.py @@ -106,6 +106,10 @@ def run( - `individual_scores` - A list of numbers from 0.0 to 1.0 that represents how high the first retrieved document is ranked. """ + if len(ground_truth_documents) == 0 or len(retrieved_documents) == 0: + msg = "ground_truth_documents and retrieved_documents must be provided." + raise ValueError(msg) + if len(ground_truth_documents) != len(retrieved_documents): msg = "The length of ground_truth_documents and retrieved_documents must be the same." raise ValueError(msg) diff --git a/haystack/components/evaluators/document_recall.py b/haystack/components/evaluators/document_recall.py index f42cccd8950..68e591d7bf5 100644 --- a/haystack/components/evaluators/document_recall.py +++ b/haystack/components/evaluators/document_recall.py @@ -172,6 +172,10 @@ def run( - `individual_scores` - A list of numbers from 0.0 to 1.0 that represents the proportion of matching documents retrieved. If the mode is `single_hit`, the individual scores are 0 or 1. """ + if len(ground_truth_documents) == 0 or len(retrieved_documents) == 0: + msg = "ground_truth_documents and retrieved_documents must be provided." + raise ValueError(msg) + if len(ground_truth_documents) != len(retrieved_documents): msg = "The length of ground_truth_documents and retrieved_documents must be the same." raise ValueError(msg) diff --git a/releasenotes/notes/evaluator-empty-input-7c2a9e4b1d8f3065.yaml b/releasenotes/notes/evaluator-empty-input-7c2a9e4b1d8f3065.yaml new file mode 100644 index 00000000000..5a6637e6875 --- /dev/null +++ b/releasenotes/notes/evaluator-empty-input-7c2a9e4b1d8f3065.yaml @@ -0,0 +1,7 @@ +--- +fixes: + - | + ``DocumentMAPEvaluator``, ``DocumentMRREvaluator``, ``DocumentRecallEvaluator`` and + ``AnswerExactMatchEvaluator`` now raise ``ValueError`` when given no questions, instead of + ``ZeroDivisionError``. This matches ``DocumentNDCGEvaluator``. A question whose own document + or answer list is empty is unchanged. diff --git a/test/components/evaluators/test_empty_input.py b/test/components/evaluators/test_empty_input.py new file mode 100644 index 00000000000..608c1873723 --- /dev/null +++ b/test/components/evaluators/test_empty_input.py @@ -0,0 +1,49 @@ +# SPDX-FileCopyrightText: 2022-present deepset GmbH +# +# SPDX-License-Identifier: Apache-2.0 + +import pytest + +from haystack import Document +from haystack.components.evaluators.answer_exact_match import AnswerExactMatchEvaluator +from haystack.components.evaluators.document_map import DocumentMAPEvaluator +from haystack.components.evaluators.document_mrr import DocumentMRREvaluator +from haystack.components.evaluators.document_recall import DocumentRecallEvaluator, RecallMode + + +@pytest.mark.parametrize( + "evaluator", + [ + DocumentMAPEvaluator(), + DocumentMRREvaluator(), + DocumentRecallEvaluator(mode=RecallMode.SINGLE_HIT), + DocumentRecallEvaluator(mode=RecallMode.MULTI_HIT), + ], +) +def test_document_evaluators_reject_top_level_empty_input(evaluator): + with pytest.raises(ValueError, match="must be provided"): + evaluator.run(ground_truth_documents=[], retrieved_documents=[]) + + with pytest.raises(ValueError, match="must be provided"): + evaluator.run(ground_truth_documents=[], retrieved_documents=[[Document(content="x")]]) + + with pytest.raises(ValueError, match="must be provided"): + evaluator.run(ground_truth_documents=[[Document(content="x")]], retrieved_documents=[]) + + +def test_document_evaluators_still_score_a_question_with_no_documents(): + ground_truth = [[]] + retrieved = [[Document(content="x")]] + for evaluator in (DocumentMAPEvaluator(), DocumentMRREvaluator(), DocumentRecallEvaluator()): + result = evaluator.run(ground_truth_documents=ground_truth, retrieved_documents=retrieved) + assert result["score"] == 0.0 + assert result["individual_scores"] == [0.0] + + +def test_answer_exact_match_rejects_top_level_empty_input(): + evaluator = AnswerExactMatchEvaluator() + with pytest.raises(ValueError, match="must be provided"): + evaluator.run(ground_truth_answers=[], predicted_answers=[]) + + result = evaluator.run(ground_truth_answers=["Berlin"], predicted_answers=["Berlin"]) + assert result == {"individual_scores": [1], "score": 1.0}