From 9fc2d9b5dff843a3b70bf9d42c298c5368f92ea9 Mon Sep 17 00:00:00 2001 From: Mohammad Hijjawi Date: Mon, 28 Sep 2026 20:08:33 +0100 Subject: [PATCH] [Fix] Compare TydiQA answers case-insensitively TydiQAEvaluator lowercased the prediction but not the reference answers, so any gold answer containing an uppercase letter could never reach exact match and lost F1 on every capitalized token, even when the model output matched the reference verbatim. Lowercase the references as well, matching the SQuAD normalization the evaluator is based on. --- opencompass/datasets/tydiqa.py | 1 + tests/datasets/test_tydiqa.py | 24 ++++++++++++++++++++++++ 2 files changed, 25 insertions(+) create mode 100644 tests/datasets/test_tydiqa.py diff --git a/opencompass/datasets/tydiqa.py b/opencompass/datasets/tydiqa.py index c27e738d9..1398b1943 100644 --- a/opencompass/datasets/tydiqa.py +++ b/opencompass/datasets/tydiqa.py @@ -75,6 +75,7 @@ def score(self, predictions, references): } for prediction, reference in zip(predictions, references): prediction = re.split(r'[\n]', prediction, 1)[0].lower() + reference = [ref.lower() for ref in reference] exact_match += self.metric_max_over_ground_truths( self.exact_match_score, prediction, reference) f1 += self.metric_max_over_ground_truths(self.f1_score, prediction, diff --git a/tests/datasets/test_tydiqa.py b/tests/datasets/test_tydiqa.py new file mode 100644 index 000000000..1dcaa8f21 --- /dev/null +++ b/tests/datasets/test_tydiqa.py @@ -0,0 +1,24 @@ +import unittest + +from opencompass.datasets.tydiqa import TydiQAEvaluator + + +class TestTydiQAEvaluator(unittest.TestCase): + + def test_score_ignores_case_of_references(self): + evaluator = TydiQAEvaluator() + result = evaluator.score( + predictions=['Paris', 'Barack Obama was president'], + references=[['Paris'], ['Barack Obama']]) + self.assertEqual(result['exact_match'], 50.0) + self.assertAlmostEqual(result['f1'], (100.0 + 200 / 3) / 2) + + def test_score_wrong_answer(self): + evaluator = TydiQAEvaluator() + result = evaluator.score(predictions=['London'], + references=[['Paris']]) + self.assertEqual(result, {'exact_match': 0.0, 'f1': 0.0}) + + +if __name__ == '__main__': + unittest.main()