From 6fd3abafb37c22255d462f3817ca3491994f7bd0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E4=BF=AE=E9=9B=A8?= <47820304+PeterGuy326@users.noreply.github.com> Date: Thu, 17 Sep 2026 17:20:20 +0800 Subject: [PATCH] fix(recall): rebuild fail-closed live producer on current main --- CHANGELOG.md | 2 + benchmarks/recall/README.md | 95 ++++++ benchmarks/recall/__main__.py | 41 +++ benchmarks/recall/live_producer.py | 261 +++++++++++++++ benchmarks/recall/tests/test_live_producer.py | 297 ++++++++++++++++++ 5 files changed, 696 insertions(+) create mode 100644 benchmarks/recall/live_producer.py create mode 100644 benchmarks/recall/tests/test_live_producer.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 9a60f1c..c72e149 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -136,6 +136,8 @@ The project publishes 0.x prerelease versions; a stable release line is not yet - The release guard suites count manifest lines without `wc -l`, whose BSD implementation pads the count with blanks, and no longer need GNU `find -printf`. +- Add an opt-in file-search ranking producer with explicit request failures, conservative result identity mapping, and operator-declared configuration. Live provider quality remains separately unverified. + - The npm installer no longer aborts a concurrent first run on Windows. The per-asset cache lock previously treated only `EEXIST` as contention, but a contended `mkdir` on Windows may raise `EPERM` or `EACCES`, so a process diff --git a/benchmarks/recall/README.md b/benchmarks/recall/README.md index 7b3cd88..69a1e10 100644 --- a/benchmarks/recall/README.md +++ b/benchmarks/recall/README.md @@ -1,5 +1,14 @@ # Multilingual recall benchmark +> **Evidence boundary: producer checks do not complete the live benchmark.** +> +> The producer corrections tracked in #196 do not complete #175's acceptance. +> The original #184 live-run requirement still needs a populated real memd, +> actual embedding-provider output and saved ranking artifacts. Unit or +> loopback HTTP fixtures establish transport behavior only. A real model-free +> lexical run also cannot establish vector/provider quality. See +> [Exact remaining live prerequisites](#exact-remaining-live-prerequisites). + This directory provides a small, repeatable retrieval benchmark. It is a decision aid for comparing lexical, vector and hybrid configurations; it is not evidence of production recall. @@ -167,3 +176,89 @@ time and sanitized query failures. It does not copy query text, corpus text, vectors or free-form provider errors. Credential-shaped configuration keys such as `api_key`, `password`, `secret`, `token` and `authorization` are rejected instead of being copied into an artifact. + +## Live memd producer + +The `produce` subcommand queries a running `memd` over file-search queries and +emits a `mem.recall-rankings.v1` file that the existing `run --rankings` path +consumes. Latency is measured client-side per request; the `0 ms` sentinel +warning above applies only to the offline lexical lane. + +```bash +python3 -m benchmarks.recall produce \ + --memd-url http://localhost:8080 \ + --token "$MEM_TOKEN" \ + --dataset benchmarks/recall/data/profile-text-v1 \ + --output /tmp/live-rankings.json \ + --dimension 768 --provider "$MEM_PROVIDER_LABEL" --model "$MEM_MODEL_LABEL" \ + --mode vector +``` + +Then score the saved rankings (the default v1 baseline uses a different corpus): + +```bash +python3 -m benchmarks.recall run \ + --dataset benchmarks/recall/data/profile-text-v1 \ + --rankings /tmp/live-rankings.json \ + --output /tmp/live-artifact.json +``` + +Load the synthetic file corpus into an isolated test deployment first. The +producer does not ingest it. Supply a token bound to the dataset's workspace; +its labels do not establish the token's real workspace identity. + +The producer maps each API result back to a dataset `doc_id` using the returned +folder `path` plus file `name`. Cross-workspace path collisions, unknown paths, +or ambiguous snippets fail the query instead of silently dropping evidence. +A same-workspace snippet-overlap tie also fails closed: document ID ordering +would invent identity evidence. Malformed paths, names, snippets and scores +fail with `invalid_result`, including malformed duplicate hits. +Query filters are translated where the API supports them: `path_prefix` becomes +`scope`, and `source_kind` becomes `type` (`image_caption` → `image`, +`text` → `text`). The `workspace` filter is not sent to the API because the +auth token determines workspace scope. A single token cannot select several +workspaces, so use a single-workspace fixture. Metadata filters are unsupported +and produce `unsupported_filter` without contacting the server; they are never +silently ignored. + +Vector mode sends `route=text`; it does not claim a hybrid lexical/vector or +multimodal experiment. Lexical mode sends `route=lexical` and requires the +server capability from #183. It emits null provider/model/dimension as required +by the ranking schema. Provider, model, dimension and index configuration are +operator declarations, not discovered or verified server metadata. The +`hardware.host` value deliberately records only the producer client's +OS/architecture, never its hostname. It is not the server's hardware inventory +and cannot establish comparable performance conditions. + +The full v1 corpus also contains structured-memory queries. `/v1/search` cannot +serve these; the producer records `unsupported_source_kind` and exits 2. Any +HTTP, mapping or response error also exits 2 while retaining an error artifact. +Use the existing `profile-text-v1` file-only fixture for this producer's bounded +acceptance. An HTTP fixture test proves transport and artifact handling only; +it does not establish live provider quality, production latency, or full-corpus +acceptance. Those remain NOT VERIFIED until a real populated memd run is saved. + + +### Exact remaining live prerequisites + +1. Use an isolated authorized memd/Worker deployment and disposable PostgreSQL + database, and verify that its token is bound to the intended test workspace. +2. Load all five synthetic files from `data/profile-text-v1/corpus.jsonl`, keeping + paths and content intact. Verify indexing and actual file identities before + interpreting the producer's output. Direct database seeding can establish + retrieval/transport behavior but does not verify ingestion or Worker indexing. +3. For vector acceptance, select the same actual embedding model for corpus and + queries, record its dimension and profile, and independently inspect the + active generation/index and deployment revision. Producer labels alone do + not verify any of those properties. +4. Run the documented producer command, retain its output, score the resulting + rankings and record errors and measured client latencies. An empty/error run + or a fake embedding provider cannot establish vector quality. A lexical run + requires the separate #194 server capability and remains a distinct result. +5. The bounded file-only experiment does not cover #175's bilingual image-query + acceptance or the full v1 structured-memory corpus. The original issue and + live quality acceptance must not be described as complete on this evidence. + +The producer remains opt-in; the normal recall CI gate runs deterministic unit +and fixture checks only. No real-model baseline is checked in until its actual +configuration and saved run are available for review. diff --git a/benchmarks/recall/__main__.py b/benchmarks/recall/__main__.py index 9fb888c..fe15870 100644 --- a/benchmarks/recall/__main__.py +++ b/benchmarks/recall/__main__.py @@ -7,7 +7,9 @@ import sys import tempfile +from .dataset import load_dataset from .errors import BenchmarkError +from .live_producer import produce_rankings from .runner import ( compare_artifacts, comparison_summary, @@ -66,6 +68,24 @@ def _parser() -> argparse.ArgumentParser: type=Path, default=PACKAGE_ROOT / "fixtures" / "external-rankings.leak.v1.json", ) + + produce = subparsers.add_parser( + "produce", + help="query a live memd and emit mem.recall-rankings.v1", + ) + produce.add_argument("--memd-url", required=True, help="base URL of memd") + produce.add_argument("--token", required=True, help="bearer token for auth") + produce.add_argument("--dataset", type=Path, default=DEFAULT_DATASET) + produce.add_argument("--output", type=Path, required=True) + produce.add_argument("--limit", type=int, default=10) + produce.add_argument("--timeout", type=float, default=30.0) + produce.add_argument("--engine", default="live-memd") + produce.add_argument("--dimension", type=int, default=768) + produce.add_argument( + "--mode", default="vector", choices=["lexical", "vector"] + ) + produce.add_argument("--provider", default="operator-unspecified") + produce.add_argument("--model", default="operator-unspecified") return parser @@ -100,6 +120,27 @@ def main(argv: list[str] | None = None) -> int: print(comparison_summary(comparison)) return 2 if candidate["metrics"]["overall"]["leakage_count"] else 0 + if args.command == "produce": + dataset = load_dataset(args.dataset) + rankings = produce_rankings( + dataset, + base_url=args.memd_url, + token=args.token, + limit=args.limit, + timeout=args.timeout, + engine_label=args.engine, + dimension=args.dimension, + mode=args.mode, + provider=args.provider, + model=args.model, + ) + write_json(args.output, rankings) + ok_count = sum(1 for q in rankings["queries"] if q["status"] == "ok") + err_count = sum(1 for q in rankings["queries"] if q["status"] == "error") + print(f"produced rankings: {ok_count} ok, {err_count} error") + print(f"rankings artifact: {args.output}") + return 2 if err_count else 0 + first = run_benchmark( dataset_dir=args.dataset, generated_at="2000-01-01T00:00:00+00:00", diff --git a/benchmarks/recall/live_producer.py b/benchmarks/recall/live_producer.py new file mode 100644 index 0000000..9888b36 --- /dev/null +++ b/benchmarks/recall/live_producer.py @@ -0,0 +1,261 @@ +"""Produce mem.recall-rankings.v1 from a live memd instance. + +Queries each dataset query against POST /v1/search, maps API results back to +dataset doc_ids by path, and emits the rankings JSON that the existing harness +consumes via --rankings. +""" + +from __future__ import annotations + +import json +import math +import platform +import posixpath +import time +import unicodedata +from typing import Any +from urllib.error import HTTPError, URLError +from urllib.request import Request, urlopen + +from .dataset import Dataset, Document +from .errors import BenchmarkError + +_SOURCE_KIND_TO_TYPE = { + "image_caption": "image", + "text": "text", +} + + +def _build_path_index(documents: list[Document]) -> dict[str, list[Document]]: + index: dict[str, list[Document]] = {} + for doc in documents: + index.setdefault(doc.path, []).append(doc) + return index + + +def _match_doc_by_path( + api_path: str, + snippet: str, + candidates: list[Document], +) -> Document | None: + if not candidates: + return None + if len(candidates) == 1: + return candidates[0] + # A snippet cannot establish tenant identity. Never choose an authorized + # document merely because a foreign document shares its path or words. + if len({doc.workspace for doc in candidates}) != 1: + return None + normalized_snippet = unicodedata.normalize("NFKC", snippet).casefold() + best: Document | None = None + best_overlap = 0 + for doc in candidates: + doc_tokens = set(unicodedata.normalize("NFKC", doc.text).casefold().split()) + overlap = sum(1 for t in normalized_snippet.split() if t in doc_tokens) + if overlap > best_overlap: + best_overlap = overlap + best = doc + elif overlap == best_overlap: + best = None + return best + + +def _source_kind_to_api_type(source_kind: str) -> str | None: + return _SOURCE_KIND_TO_TYPE.get(source_kind) + + +def _coarse_host() -> str: + """Record OS/architecture only, never a hostname or client identity.""" + try: + return f"{platform.system()}/{platform.machine()}" + except Exception: + return "unknown" + + +def _query_memd( + base_url: str, + token: str, + query_text: str, + *, + scope: str = "", + type_filter: str = "", + route: str = "auto", + limit: int = 10, + timeout: float = 30.0, +) -> tuple[list[dict[str, Any]], float, str | None]: + body: dict[str, Any] = {"query": query_text, "limit": limit, "route": route} + if scope: + body["scope"] = scope + if type_filter: + body["type"] = type_filter + + url = base_url.rstrip("/") + "/v1/search" + data = json.dumps(body).encode("utf-8") + req = Request(url, data=data, method="POST") + req.add_header("Content-Type", "application/json") + req.add_header("Authorization", f"Bearer {token}") + + start = time.perf_counter() + try: + with urlopen(req, timeout=timeout) as resp: + payload = json.loads(resp.read().decode("utf-8")) + elapsed_ms = (time.perf_counter() - start) * 1000.0 + if not isinstance(payload, dict) or "results" not in payload: + return [], elapsed_ms, "invalid_response" + results = payload["results"] + if results is None: + results = [] # memd encodes an empty nil hit slice as null. + if not isinstance(results, list) or any(not isinstance(hit, dict) for hit in results): + return [], elapsed_ms, "invalid_response" + return results, elapsed_ms, None + except HTTPError as exc: + elapsed_ms = (time.perf_counter() - start) * 1000.0 + return [], elapsed_ms, f"http_{exc.code}" + except URLError: + elapsed_ms = (time.perf_counter() - start) * 1000.0 + return [], elapsed_ms, "connection_error" + except Exception: + elapsed_ms = (time.perf_counter() - start) * 1000.0 + return [], elapsed_ms, "unknown_error" + + +def produce_rankings( + dataset: Dataset, + *, + base_url: str, + token: str, + limit: int = 10, + timeout: float = 30.0, + engine_label: str = "live-memd", + dimension: int = 768, + mode: str = "vector", + provider: str = "operator-unspecified", + model: str = "operator-unspecified", +) -> dict[str, Any]: + if mode not in {"lexical", "vector"}: + raise BenchmarkError("memd /v1/search does not expose a hybrid lexical/vector route") + if not 1 <= limit <= 100 or not math.isfinite(timeout) or timeout <= 0: + raise BenchmarkError("limit must be 1..100 and timeout must be positive and finite") + if not isinstance(engine_label, str) or not engine_label.strip() or engine_label == "lexical-reference": + raise BenchmarkError("engine must identify live memd, not lexical-reference") + if mode == "vector" and ( + not isinstance(dimension, int) or isinstance(dimension, bool) or dimension <= 0 + or not isinstance(provider, str) or not provider.strip() + or not isinstance(model, str) or not model.strip() + ): + raise BenchmarkError("vector mode requires a positive dimension and non-empty provider/model labels") + path_index = _build_path_index(list(dataset.documents)) + + query_rows: list[dict[str, Any]] = [] + for query in dataset.queries: + if query.expected_source_kind == "structured": + query_rows.append({"query_id": query.id, "status": "error", + "latency_ms": 0.0, "results": [], + "error_code": "unsupported_source_kind"}) + continue + if query.filters.get("metadata"): + query_rows.append({"query_id": query.id, "status": "error", + "latency_ms": 0.0, "results": [], + "error_code": "unsupported_filter"}) + continue + scope = query.filters.get("path_prefix", "") + type_filter = _source_kind_to_api_type(query.expected_source_kind) or "" + + api_results, latency_ms, error_code = _query_memd( + base_url, + token, + query.text, + scope=scope, + type_filter=type_filter, + route="lexical" if mode == "lexical" else "text", + limit=limit, + timeout=timeout, + ) + + if error_code: + row: dict[str, Any] = { + "query_id": query.id, + "status": "error", + "latency_ms": round(latency_ms, 2), + "results": [], + "error_code": error_code, + } + query_rows.append(row) + continue + + mapped_results: list[dict[str, Any]] = [] + seen_doc_ids: set[str] = set() + mapping_error: str | None = None + for hit in api_results: + hit_path = hit.get("path") + name = hit.get("name", "") + snippet = hit.get("snippet", "") + score = hit.get("score") + # Validate every row before deduplication: a malformed duplicate is + # still evidence of a failed response, not something to discard. + if ( + not isinstance(hit_path, str) or not hit_path.startswith("/") + or not isinstance(name, str) or not isinstance(snippet, str) + or (name and ("/" in name or name in {".", ".."})) + or (score is not None and ( + isinstance(score, bool) or not isinstance(score, (int, float)) + or not math.isfinite(score) + )) + ): + mapping_error = "invalid_result" + break + if name: + # memd returns a folder path and file name separately. + hit_path = posixpath.join(hit_path, name) + candidates = path_index.get(hit_path, []) + doc = _match_doc_by_path(hit_path, snippet, candidates) + if doc is None: + mapping_error = "unmapped_result" + break + if doc.id in seen_doc_ids: + continue + seen_doc_ids.add(doc.id) + result: dict[str, Any] = { + "doc_id": doc.id, + "citation": doc.citation, + } + if score is not None: + result["score"] = float(score) + mapped_results.append(result) + + if mapping_error: + mapped_results = [] + status = "ok" if not mapping_error else "error" + row = { + "query_id": query.id, + "status": status, + "latency_ms": round(latency_ms, 2), + "results": mapped_results, + } + if mapping_error: + row["error_code"] = mapping_error + query_rows.append(row) + + return { + "schema_version": "mem.recall-rankings.v1", + "engine": engine_label, + "configuration": { + "mode": mode, + "provider": None if mode == "lexical" else provider, + "model": None if mode == "lexical" else model, + "dimension": None if mode == "lexical" else dimension, + "evidence": "operator-declared configuration; model and index not verified by producer", + "index": { + "kind": "operator-unspecified", + }, + "search": { + "top_k": limit, + "route": "lexical" if mode == "lexical" else "text", + "workspace": "bound by the supplied token; not inferred from dataset labels", + }, + }, + "hardware": { + "host": _coarse_host(), + }, + "queries": query_rows, + } diff --git a/benchmarks/recall/tests/test_live_producer.py b/benchmarks/recall/tests/test_live_producer.py new file mode 100644 index 0000000..6985e7e --- /dev/null +++ b/benchmarks/recall/tests/test_live_producer.py @@ -0,0 +1,297 @@ +from __future__ import annotations + +import json +from http.server import BaseHTTPRequestHandler, HTTPServer +from pathlib import Path +import tempfile +import threading +import unittest +from unittest.mock import patch + +from benchmarks.recall.__main__ import main +from benchmarks.recall.adapters import load_external_rankings +from benchmarks.recall.errors import BenchmarkError +from benchmarks.recall.live_producer import ( + _build_path_index, + _match_doc_by_path, + produce_rankings, +) +from benchmarks.recall.dataset import Document, load_dataset + + +class PathIndexTest(unittest.TestCase): + def test_index_groups_by_path(self) -> None: + docs = [ + Document( + id="a", language="en", source_kind="text", workspace="alpha", + path="/notes/a.md", citation="mem://files/a", + text="alpha note", metadata={}, + ), + Document( + id="b", language="en", source_kind="text", workspace="alpha", + path="/notes/a.md", citation="mem://files/b", + text="beta note", metadata={}, + ), + ] + index = _build_path_index(docs) + self.assertEqual(len(index["/notes/a.md"]), 2) + + def test_match_single_candidate(self) -> None: + doc = Document( + id="solo", language="en", source_kind="text", workspace="alpha", + path="/notes/solo.md", citation="mem://files/solo", + text="unique content", metadata={}, + ) + result = _match_doc_by_path("/notes/solo.md", "anything", [doc]) + self.assertEqual(result, doc) + + def test_match_picks_best_snippet_overlap(self) -> None: + doc_a = Document( + id="a", language="en", source_kind="text", workspace="alpha", + path="/notes/shared.md", citation="mem://files/a", + text="saturn ring observation", metadata={}, + ) + doc_b = Document( + id="b", language="en", source_kind="text", workspace="alpha", + path="/notes/shared.md", citation="mem://files/b", + text="completely different topic", metadata={}, + ) + result = _match_doc_by_path("/notes/shared.md", "saturn ring", [doc_a, doc_b]) + self.assertEqual(result, doc_a) + + def test_ambiguous_path_does_not_guess_identity(self) -> None: + docs = [Document(id=key, language="en", source_kind="text", + workspace=workspace, path="/shared.md", citation="mem://"+key, + text="same snippet", metadata={}) + for key, workspace in [("a", "alpha"), ("b", "beta")]] + self.assertIsNone(_match_doc_by_path("/shared.md", "same snippet", docs)) + + +class ProduceRankingsTest(unittest.TestCase): + def setUp(self) -> None: + self.tempdir = tempfile.TemporaryDirectory() + self.root = Path(self.tempdir.name) + self.dataset = self.root / "dataset" + self.dataset.mkdir() + (self.dataset / "dataset.json").write_text( + json.dumps({ + "schema_version": "mem.recall-dataset.v1", + "version": "unit-test-v1", + "provenance": "hand-authored synthetic data", + "license": "CC0-1.0", + "required_coverage": { + "slices": ["exact"], "languages": ["en"], "source_kinds": ["text"], + }, + }), + encoding="utf-8", + ) + (self.dataset / "corpus.jsonl").write_text( + json.dumps({ + "id": "file-en-cassini", "language": "en", "source_kind": "text", + "workspace": "alpha", "path": "/research/saturn.md", + "citation": "mem://files/file-en-cassini", + "text": "Cassini observed Saturn hexagonal storm", + "provenance": "synthetic", + }) + "\n", + encoding="utf-8", + ) + (self.dataset / "queries.jsonl").write_text( + json.dumps({ + "id": "q-en-text-exact", "text": "Cassini Saturn hexagonal storm", + "language": "en", "slice": "exact", + "filters": {"workspace": "alpha", "source_kind": "text"}, + "expected_source_kind": "text", + }) + "\n", + encoding="utf-8", + ) + (self.dataset / "qrels.json").write_text( + json.dumps({"q-en-text-exact": {"file-en-cassini": 3}}), + encoding="utf-8", + ) + + def tearDown(self) -> None: + self.tempdir.cleanup() + + def test_http_fixture_runs_transport_and_emits_loadable_artifact(self) -> None: + # Real loopback HTTP, synthetic response: this is not a live memd or + # embedding-quality benchmark. + requests = [] + + class Handler(BaseHTTPRequestHandler): + def do_POST(self): + requests.append((self.path, json.loads(self.rfile.read(int(self.headers["Content-Length"]))))) + payload = json.dumps({"results": [{"path": "/research", "name": "saturn.md", "score": 0.9}]}).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(payload))) + self.end_headers() + self.wfile.write(payload) + + def log_message(self, *args): + pass + + server = HTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + output = self.root / "http-fixture.json" + try: + code = main(["produce", "--memd-url", f"http://127.0.0.1:{server.server_port}", + "--token", "fixture", "--dataset", str(self.dataset), + "--output", str(output), "--provider", "fixture", "--model", "fixture"]) + finally: + server.shutdown() + thread.join() + server.server_close() + self.assertEqual(code, 0) + self.assertEqual(requests[0][0], "/v1/search") + self.assertEqual(requests[0][1]["route"], "text") + artifact = load_external_rankings(output, load_dataset(self.dataset)) + self.assertEqual(artifact["queries"][0]["results"][0]["doc_id"], "file-en-cassini") + self.assertGreater(artifact["queries"][0]["latency_ms"], 0) + self.assertNotIn("client", artifact["hardware"]) + + @patch("benchmarks.recall.live_producer._query_memd") + def test_produce_rankings_success(self, mock_query: unittest.mock.MagicMock) -> None: + mock_query.return_value = ( + [{"path": "/research/saturn.md", "snippet": "Cassini observed Saturn hexagonal storm", "score": 0.95}], + 12.5, None, + ) + dataset = load_dataset(self.dataset) + rankings = produce_rankings( + dataset, base_url="http://localhost:8080", token="test-token", dimension=768, + ) + self.assertEqual(rankings["schema_version"], "mem.recall-rankings.v1") + self.assertEqual(rankings["engine"], "live-memd") + self.assertEqual(rankings["configuration"]["dimension"], 768) + self.assertEqual(len(rankings["queries"]), 1) + query_row = rankings["queries"][0] + self.assertEqual(query_row["query_id"], "q-en-text-exact") + self.assertEqual(query_row["status"], "ok") + self.assertGreater(query_row["latency_ms"], 0) + self.assertEqual(len(query_row["results"]), 1) + self.assertEqual(query_row["results"][0]["doc_id"], "file-en-cassini") + + @patch("benchmarks.recall.live_producer._query_memd") + def test_produce_rankings_error(self, mock_query: unittest.mock.MagicMock) -> None: + mock_query.return_value = ([], 5.0, "http_503") + dataset = load_dataset(self.dataset) + rankings = produce_rankings( + dataset, base_url="http://localhost:8080", token="test-token", + ) + query_row = rankings["queries"][0] + self.assertEqual(query_row["status"], "error") + self.assertEqual(query_row["error_code"], "http_503") + self.assertEqual(query_row["results"], []) + + @patch("benchmarks.recall.live_producer._query_memd") + def test_maps_shipping_folder_and_name_shape(self, mock_query) -> None: + mock_query.return_value = ([{"path": "/research", "name": "saturn.md", "score": 0.9}], 1, None) + rankings = produce_rankings(load_dataset(self.dataset), base_url="http://localhost", token="test") + self.assertEqual(rankings["queries"][0]["results"][0]["doc_id"], "file-en-cassini") + + @patch("benchmarks.recall.live_producer._query_memd") + def test_unmapped_hit_is_not_silently_discarded(self, mock_query) -> None: + mock_query.return_value = ([{"path": "/foreign.md", "snippet": "private"}], 1, None) + rankings = produce_rankings(load_dataset(self.dataset), base_url="http://localhost", token="test") + self.assertEqual(rankings["queries"][0]["status"], "error") + self.assertEqual(rankings["queries"][0]["error_code"], "unmapped_result") + + @patch("benchmarks.recall.live_producer._query_memd") + def test_lexical_artifact_is_loadable_and_route_is_sent(self, mock_query) -> None: + mock_query.return_value = ([], 1, None) + dataset = load_dataset(self.dataset) + rankings = produce_rankings(dataset, base_url="http://localhost", token="test", mode="lexical") + output = self.root / "rankings.json" + output.write_text(json.dumps(rankings), encoding="utf-8") + load_external_rankings(output, dataset) + self.assertEqual(mock_query.call_args.kwargs["route"], "lexical") + + @patch("benchmarks.recall.live_producer._query_memd") + def test_produce_command_fails_when_requests_fail(self, mock_query) -> None: + mock_query.return_value = ([], 1, "http_503") + code = main(["produce", "--memd-url", "http://localhost", "--token", "test", + "--dataset", str(self.dataset), "--output", str(self.root / "failed.json")]) + self.assertEqual(code, 2) + + + @patch("benchmarks.recall.live_producer._query_memd") + def test_malformed_hits_retain_error_artifact(self, mock_query) -> None: + good = {"path": "/research", "name": "saturn.md", "score": 0.9} + malformed = [ + {"path": None}, {"path": []}, {"path": {}}, {"path": 42}, + {"path": "/research", "name": None}, + {"path": "/research", "name": "/research/saturn.md"}, + {"path": "/research", "name": "../saturn.md"}, + {"path": "/research/saturn.md", "snippet": []}, + {"path": "/research/saturn.md", "score": False}, + {"path": "/research/saturn.md", "score": "0.9"}, + {"path": "/research/saturn.md", "score": float("nan")}, + {"path": "/research/saturn.md", "score": float("inf")}, + ] + for hit in malformed: + with self.subTest(hit=hit): + # Including a valid first row also checks that malformed + # duplicate rows cannot disappear during deduplication. + mock_query.return_value = ([good, hit], 1, None) + output = self.root / "malformed.json" + code = main(["produce", "--memd-url", "http://localhost", "--token", "fixture", + "--dataset", str(self.dataset), "--output", str(output)]) + self.assertEqual(code, 2) + row = load_external_rankings(output, load_dataset(self.dataset))["queries"][0] + self.assertEqual(row["error_code"], "invalid_result") + self.assertEqual(row["results"], []) + + @patch("benchmarks.recall.live_producer._query_memd") + def test_unsupported_metadata_filter_never_contacts_server(self, mock_query) -> None: + path = self.dataset / "queries.jsonl" + query = json.loads(path.read_text()) + query["filters"]["metadata"] = {"project": "saturn"} + path.write_text(json.dumps(query) + "\n") + # Give the relevant document the same metadata so dataset validation + # succeeds; the unsupported transport filter remains the only failure. + path = self.dataset / "corpus.jsonl" + document = json.loads(path.read_text()) + document["metadata"] = {"project": "saturn"} + path.write_text(json.dumps(document) + "\n") + rankings = produce_rankings(load_dataset(self.dataset), base_url="http://localhost", token="fixture") + self.assertEqual(rankings["queries"][0]["error_code"], "unsupported_filter") + mock_query.assert_not_called() + + @patch("benchmarks.recall.live_producer._query_memd") + def test_structured_queries_never_contact_server(self, mock_query) -> None: + from dataclasses import replace + dataset = load_dataset(self.dataset) + dataset = replace(dataset, queries=(replace(dataset.queries[0], expected_source_kind="structured"),)) + rankings = produce_rankings(dataset, base_url="http://localhost", token="fixture") + self.assertEqual(rankings["queries"][0]["error_code"], "unsupported_source_kind") + mock_query.assert_not_called() + + @patch("benchmarks.recall.live_producer._query_memd") + def test_invalid_configuration_never_contacts_server(self, mock_query) -> None: + for kwargs in [ + {"dimension": 0}, {"dimension": -1}, {"dimension": True}, + {"provider": ""}, {"model": " "}, {"engine_label": "lexical-reference"}, + {"limit": 0}, {"limit": 101}, {"timeout": float("nan")}, + {"timeout": float("inf")}, {"timeout": 0}, {"mode": "hybrid"}, + ]: + with self.subTest(kwargs=kwargs), self.assertRaises(BenchmarkError): + produce_rankings(load_dataset(self.dataset), base_url="http://localhost", token="fixture", **kwargs) + mock_query.assert_not_called() + + @patch("benchmarks.recall.live_producer._query_memd") + def test_duplicate_valid_chunks_map_to_one_document(self, mock_query) -> None: + mock_query.return_value = ([{"path": "/research/saturn.md", "score": 0.9}, + {"path": "/research/saturn.md", "score": 0.8}], 1, None) + rankings = produce_rankings(load_dataset(self.dataset), base_url="http://localhost", token="fixture") + self.assertEqual(rankings["queries"][0]["status"], "ok") + self.assertEqual(len(rankings["queries"][0]["results"]), 1) + + def test_equal_same_workspace_candidates_fail_closed(self) -> None: + from dataclasses import replace + document = load_dataset(self.dataset).documents[0] + other = replace(document, id="other") + self.assertIsNone(_match_doc_by_path(document.path, document.text, [document, other])) + + +if __name__ == "__main__": + unittest.main()