From 27e40092a926645addfcf5efa7efcf7af11c93f9 Mon Sep 17 00:00:00 2001 From: Yue Sun Date: Wed, 9 Sep 2026 15:29:30 +0800 Subject: [PATCH] fix(eval): respect image-to-text pipeline call contracts --- docs/commands/eval.md | 2 + .../modelkit/eval/image_to_text_evaluator.py | 11 +++- .../unit/eval/test_image_to_text_evaluator.py | 59 ++++++++++++++++++- 3 files changed, 69 insertions(+), 3 deletions(-) diff --git a/docs/commands/eval.md b/docs/commands/eval.md index accda82df..103d4431a 100644 --- a/docs/commands/eval.md +++ b/docs/commands/eval.md @@ -54,6 +54,8 @@ Python callers can pass an existing model directly with `evaluate(config, pytorc PyTorch text-generation evaluation adapts the checkpoint to the existing causal-LM evaluator contract and reports perplexity without ONNX export. Pre-built ONNX files, composite `role=path` models, GenAI bundles, compare mode, references, and tensor input archives remain on their existing WinML paths. `--runtime pytorch` rejects those forms, along with ONNX build, export, EP, precision, quantization, optimization, analysis, and cache-related options. +For `image-to-text`, evaluation supplies an empty text prompt only when the active pipeline explicitly accepts it. Individual pipeline failures are logged and counted as skipped samples, but if pipeline errors leave no evaluated predictions, evaluation fails with the underlying error instead of reporting null CER and CIDEr values. + ## Examples Evaluate a HuggingFace model using the task-default dataset: diff --git a/src/winml/modelkit/eval/image_to_text_evaluator.py b/src/winml/modelkit/eval/image_to_text_evaluator.py index db63404ed..965201e14 100644 --- a/src/winml/modelkit/eval/image_to_text_evaluator.py +++ b/src/winml/modelkit/eval/image_to_text_evaluator.py @@ -21,6 +21,7 @@ from __future__ import annotations +import inspect import logging from typing import TYPE_CHECKING, Any @@ -64,6 +65,9 @@ def compute(self) -> dict[str, Any]: metric = TextSimilarityMetric() skipped = 0 + last_error: Exception | None = None + # Multimodal pipelines require text; legacy image-only pipelines reject it. + call_kwargs = {"text": ""} if "text" in inspect.signature(self.pipe).parameters else {} for sample in tqdm(self.data, desc="Evaluating", unit="sample"): image = sample.get(self._image_col) @@ -73,9 +77,10 @@ def compute(self) -> dict[str, Any]: continue try: - out = self.pipe(image, text="") + out = self.pipe(image, **call_kwargs) except Exception as e: logger.warning("Pipeline call failed (skipping): %s", e) + last_error = e skipped += 1 continue @@ -91,6 +96,10 @@ def compute(self) -> dict[str, Any]: metric.update(pred.strip(), references) result = metric.compute() + if result["n_samples"] == 0 and last_error is not None: + raise RuntimeError( + f"Image-to-text evaluation produced no predictions: {last_error}" + ) from last_error if skipped: result["skipped"] = skipped return result diff --git a/tests/unit/eval/test_image_to_text_evaluator.py b/tests/unit/eval/test_image_to_text_evaluator.py index a1f0b5d03..d0dfabd59 100644 --- a/tests/unit/eval/test_image_to_text_evaluator.py +++ b/tests/unit/eval/test_image_to_text_evaluator.py @@ -7,9 +7,14 @@ from __future__ import annotations +from types import SimpleNamespace from unittest.mock import MagicMock, patch -from winml.modelkit.eval.image_to_text_evaluator import WinMLImageToTextEvaluator +import pytest +from PIL import Image +from transformers.pipelines import ImageTextToTextPipeline, ImageToTextPipeline + +from winml.modelkit.eval import WinMLImageToTextEvaluator from winml.modelkit.inference.pipeline import _HF_PIPELINE_TASK_MAP @@ -110,11 +115,47 @@ def test_perfect_predictions(self): result = ev.compute() - ev.pipe.assert_any_call("img1", text="") + ev.pipe.assert_any_call("img1") assert result["cer"] == 0.0 assert result["n_samples"] == 2 assert "cider" in result + @pytest.mark.parametrize("pipeline_class", [ImageToTextPipeline, ImageTextToTextPipeline]) + def test_real_pipeline_call_contract(self, pipeline_class): + """Exercise real argument validation without loading model weights.""" + pipe = object.__new__(pipeline_class) + pipe._num_workers = None + pipe._batch_size = None + pipe._preprocess_params = {} + pipe._forward_params = {} + pipe._postprocess_params = {} + pipe.call_count = 0 + pipe.framework = "pt" + pipe.assistant_model = None + pipe.assistant_tokenizer = None + pipe.processor = SimpleNamespace(chat_template=None) + + def predict(inputs, *_args): + if isinstance(inputs, dict): + assert inputs["text"] == "" + image = inputs["images"] + else: + image = inputs + return [{"generated_text": str(image.getpixel((0, 0)))}] + + pipe.run_single = MagicMock(side_effect=predict) + ev = make_evaluator() + images = [Image.new("RGB", (2, 2), color=(index, 0, 0)) for index in range(2)] + ev.data = [{"image": image, "text": str(image.getpixel((0, 0)))} for image in images] + ev.pipe = pipe + + result = ev.compute() + + assert pipe.run_single.call_count == len(images) + assert result["n_samples"] == len(images) + assert result["cer"] == 0.0 + assert "skipped" not in result + def test_dict_output_shape(self): """Pipeline may also return a single dict (not a list).""" ev = make_evaluator() @@ -168,6 +209,20 @@ def test_pipeline_exception_skipped(self): assert result["cer"] == 0.0 assert result.get("skipped") == 1 + @pytest.mark.parametrize("error_class", [TypeError, RuntimeError]) + def test_all_pipeline_exceptions_raise(self, error_class): + """A broken pipeline must not return successful empty metrics.""" + ev = make_evaluator() + ev.data = [{"image": f"image-{index}", "text": str(index)} for index in range(2)] + error = error_class("pipeline failed") + ev.pipe = MagicMock(side_effect=error) + + with pytest.raises(RuntimeError, match="produced no predictions: pipeline failed") as exc: + ev.compute() + + assert exc.value.__cause__ is error + assert ev.pipe.call_count == len(ev.data) + def test_uses_custom_columns(self): """Image and label columns from columns_mapping are honoured.""" ev = make_evaluator(