From b5a2eb3970dcd0abf816800216227bfcd45bd3e0 Mon Sep 17 00:00:00 2001 From: Igor Apresov Date: Thu, 24 Sep 2026 21:44:25 +0300 Subject: [PATCH 1/4] feat(mcp): page ranked search results with stable cursor --- runtime/v8std_mcp_index.py | 103 +++++++++++++++++++++-------- runtime/v8std_mcp_runtime.py | 7 +- runtime/v8std_mcp_server.py | 7 +- tests/test_v8std_mcp_index.py | 53 +++++++++++++++ tests/test_v8std_mcp_runtime.py | 1 + tests/test_v8std_mcp_tools_only.py | 23 ++++++- 6 files changed, 162 insertions(+), 32 deletions(-) diff --git a/runtime/v8std_mcp_index.py b/runtime/v8std_mcp_index.py index 7c234d5..622ffdf 100644 --- a/runtime/v8std_mcp_index.py +++ b/runtime/v8std_mcp_index.py @@ -36,6 +36,7 @@ MAX_ID_OR_ALIAS_CHARS = 1000 MAX_LIMIT = 50 DEFAULT_LIMIT = 10 +SEARCH_CURSOR_RE = re.compile(r"^vs1\.([0-9a-f]{64})\.([1-9][0-9]*)$") MAX_BODY_CHARS = 12000 MAX_BODY_LIMIT_CHARS = 30000 MAX_SNIPPET_CHARS = 4000 @@ -491,42 +492,50 @@ def search( types: list[str] | None = None, mode: str = "hybrid", limit: int | None = None, + cursor: str | None = None, ) -> dict[str, Any]: query = require_text(query, "query", MAX_QUERY_CHARS) requested_limit = clamp_limit(limit) allowed_types = self._validate_types(types) mode = self._validate_mode(mode) + cursor_match = self.validate_search_cursor(cursor) self.refresh_if_needed() normalized_query = normalize_query(query) if not normalized_query: + if cursor_match is not None: + raise ValueError("invalid search cursor") return { "query": query, "normalized_query": normalized_query, "mode": mode, "types": sorted(allowed_types) if allowed_types else None, "results": [], + "total": 0, + "next_cursor": None, } - query_tokens = self._query_tokens(query) - candidates: dict[str, dict[str, Any]] = {} + # A refresh must not replace the index between ranking, fingerprinting, + # and projection of one page. The lock is reentrant for lookup helpers. + with self._lock: + query_tokens = self._query_tokens(query) + candidates: dict[str, dict[str, Any]] = {} - if mode in {"hybrid", "exact"}: - self._add_exact_scores(candidates, query, normalized_query) - self._add_code_lookup_scores(candidates, query) - self._add_fuzzy_code_scores(candidates, query) + if mode in {"hybrid", "exact"}: + self._add_exact_scores(candidates, query, normalized_query) + self._add_code_lookup_scores(candidates, query) + self._add_fuzzy_code_scores(candidates, query) - if mode in {"hybrid", "bm25"}: - self._add_bm25_scores(candidates, query_tokens) - self._add_metadata_coverage_scores(candidates, query) + if mode in {"hybrid", "bm25"}: + self._add_bm25_scores(candidates, query_tokens) + self._add_metadata_coverage_scores(candidates, query) - if mode in {"hybrid", "semantic"}: - self._add_semantic_scores(candidates, query) + if mode in {"hybrid", "semantic"}: + self._add_semantic_scores(candidates, query) - if mode == "hybrid": - self._add_related_boosts(candidates) + if mode == "hybrid": + self._add_related_boosts(candidates) - entries = [] - with self._lock: + entries = [] for page_id, candidate in candidates.items(): page = self._pages_by_id.get(page_id) if not page: @@ -541,18 +550,58 @@ def search( continue entries.append((score, page, candidate)) - entries.sort(key=lambda item: (-item[0], concrete_rank(item[1]), item[1]["type"], item[1]["id"])) - return { - "query": query, - "normalized_query": normalized_query, - "mode": mode, - "types": sorted(allowed_types) if allowed_types else None, - "semantic_enabled": self._vector_metadata is not None and bool(self._vectors), - "results": [ - self._search_entry(page, score, candidate) - for score, page, candidate in entries[:requested_limit] - ], - } + entries.sort(key=lambda item: (-item[0], concrete_rank(item[1]), item[1]["type"], item[1]["id"])) + result_fingerprint = hashlib.sha256() + result_fingerprint.update(b"v8std-search-v1\0") + result_fingerprint.update(self._metadata.sha256.encode("ascii") if self._metadata else b"") + result_fingerprint.update(b"\0") + result_fingerprint.update( + self._vector_metadata.sha256.encode("ascii") if self._vector_metadata else b"" + ) + result_fingerprint.update(b"\0") + result_fingerprint.update(json.dumps( + [query, mode, sorted(allowed_types) if allowed_types else None, requested_limit], + ensure_ascii=False, separators=(",", ":"), + ).encode("utf-8")) + for score, page, candidate in entries: + result_fingerprint.update(b"\0") + result_fingerprint.update(json.dumps( + self._search_entry(page, score, candidate), + ensure_ascii=False, sort_keys=True, separators=(",", ":"), + ).encode("utf-8")) + fingerprint = result_fingerprint.hexdigest() + offset = 0 + if cursor_match is not None: + if cursor_match.group(1) != fingerprint: + raise ValueError("stale search cursor") + offset = int(cursor_match.group(2)) + if offset >= len(entries): + raise ValueError("invalid search cursor") + next_offset = min(offset + requested_limit, len(entries)) + return { + "query": query, + "normalized_query": normalized_query, + "mode": mode, + "types": sorted(allowed_types) if allowed_types else None, + "semantic_enabled": self._vector_metadata is not None and bool(self._vectors), + "results": [ + self._search_entry(page, score, candidate) + for score, page, candidate in entries[offset:next_offset] + ], + "total": len(entries), + "next_cursor": f"vs1.{fingerprint}.{next_offset}" if next_offset < len(entries) else None, + } + + @staticmethod + def validate_search_cursor(cursor: str | None) -> re.Match[str] | None: + if cursor is None: + return None + if not isinstance(cursor, str) or len(cursor) > 128: + raise ValueError("invalid search cursor") + match = SEARCH_CURSOR_RE.fullmatch(cursor) + if match is None: + raise ValueError("invalid search cursor") + return match def page(self, id_or_alias_or_url: str, *, body_limit: int = MAX_BODY_CHARS) -> dict[str, Any]: id_or_alias_or_url = require_text( diff --git a/runtime/v8std_mcp_runtime.py b/runtime/v8std_mcp_runtime.py index aa1be7e..ff13cd2 100644 --- a/runtime/v8std_mcp_runtime.py +++ b/runtime/v8std_mcp_runtime.py @@ -83,13 +83,16 @@ def _present(self, generation, result): def _lookup(self, generation, value): return LinkCatalog(generation.canonical_site_url, self.site_url, generation.page_paths).lookup(value) - def search(self, query, *, types=None, mode="hybrid", limit=None): + def search(self, query, *, types=None, mode="hybrid", limit=None, cursor=None): require_text(query, "query", MAX_QUERY_CHARS) clamp_limit(limit) V8StdIndex._validate_types(types) V8StdIndex._validate_mode(mode) + V8StdIndex.validate_search_cursor(cursor) generation = self._current() - return self._present(generation, generation.index.search(query, types=types, mode=mode, limit=limit)) + return self._present(generation, generation.index.search( + query, types=types, mode=mode, limit=limit, cursor=cursor, + )) def page(self, id_or_alias_or_url, *, body_limit=MAX_BODY_CHARS): require_text(id_or_alias_or_url, "id_or_alias_or_url", MAX_ID_OR_ALIAS_CHARS) diff --git a/runtime/v8std_mcp_server.py b/runtime/v8std_mcp_server.py index 31070b0..10027ff 100644 --- a/runtime/v8std_mcp_server.py +++ b/runtime/v8std_mcp_server.py @@ -581,7 +581,9 @@ def build_server( 'identifier. Returns ranked page IDs, titles, descriptions, URLs, scores and match ' 'reasons; no full article text. Known diagnostic codes are handled by ' 'v8std_explain_diagnostics; BSL/SDBL source fragments by v8std_explain_snippet. An exact ' - 'page ID or URL can be read with v8std_get_page. An empty result means no match in this ' + 'page ID or URL can be read with v8std_get_page. Pass next_cursor back with the same ' + 'query, limit, types and mode to continue; an updated ranking rejects the old cursor. ' + 'An empty result means no match in this ' 'corpus, not that the code is correct. Scores rank candidates and are not probabilities.' ), ) @@ -590,8 +592,9 @@ def search( limit: Annotated[int, Field(description="Maximum results, default 10; clamped to 1–50.")] = 10, types: Annotated[list[str] | None, Field(description="Page types: standard, diagnostic, fix, pattern, service. null or [] means all types.")] = None, mode: Annotated[str, Field(description="hybrid: combined search (default); exact: includes identifier variants and fuzzy code matches; bm25: text/metadata; semantic: indexed vectors.")] = "hybrid", + cursor: Annotated[str | None, Field(description="Opaque next_cursor from a previous search with identical query, limit, types and mode; rejects changed rankings.")] = None, ) -> dict[str, Any]: - result = index.search(query, types=types, mode=mode, limit=limit) + result = index.search(query, types=types, mode=mode, limit=limit, cursor=cursor) tool_usage.record_search(query, result, system=current_client_system()) return result diff --git a/tests/test_v8std_mcp_index.py b/tests/test_v8std_mcp_index.py index ae445f9..6c217c3 100644 --- a/tests/test_v8std_mcp_index.py +++ b/tests/test_v8std_mcp_index.py @@ -56,6 +56,59 @@ def test_hybrid_search_finds_standards_and_diagnostics(self): self.assertIn("match_reasons", std_results[0]) self.assertIn("score_details", std_results[0]) + def test_search_cursor_reads_past_fifty_without_reordering_or_duplication(self): + first = self.index.search("модуль", limit=50) + self.assertGreater(first["total"], 50) + self.assertEqual(len(first["results"]), 50) + self.assertIsNotNone(first["next_cursor"]) + second = self.index.search("модуль", limit=50, cursor=first["next_cursor"]) + self.assertEqual(second, self.index.search("модуль", limit=50, cursor=first["next_cursor"])) + self.assertFalse({item["id"] for item in first["results"]} & + {item["id"] for item in second["results"]}) + with self.assertRaisesRegex(ValueError, "stale search cursor"): + self.index.search("форма", limit=50, cursor=first["next_cursor"]) + with self.assertRaisesRegex(ValueError, "stale search cursor"): + self.index.search("модуль", limit=20, cursor=first["next_cursor"]) + with self.assertRaisesRegex(ValueError, "stale search cursor"): + self.index.search("модуль", limit=50, mode="bm25", cursor=first["next_cursor"]) + with self.assertRaisesRegex(ValueError, "stale search cursor"): + self.index.search("модуль", limit=50, types=["standard"], cursor=first["next_cursor"]) + + seen = first["results"] + second["results"] + page = second + while page["next_cursor"] is not None: + page = self.index.search("модуль", limit=50, cursor=page["next_cursor"]) + seen.extend(page["results"]) + self.assertEqual(len(seen), first["total"]) + self.assertEqual(len({item["id"] for item in seen}), first["total"]) + self.assertLessEqual(len(page["results"]), 50) + + def test_search_cursor_rejects_a_changed_index(self): + with tempfile.TemporaryDirectory() as temp_dir: + pages_path = Path(temp_dir) / "pages.jsonl" + pages_path.write_text( + '\n'.join(json.dumps({"id": f"std{number}", "type": "standard", "title": "Page module"}) + for number in (1, 2, 3)) + '\n', + encoding="utf-8", + ) + index = V8StdIndex( + pages_path=pages_path, + vectors_path=REPO_ROOT / "docs" / "ai" / "search-vectors.jsonl", + ) + index.load() + first = index.search("module", limit=1) + self.assertIsNotNone(first["next_cursor"]) + pages_path.write_text( + '\n'.join(json.dumps({"id": f"std{number}", "type": "standard", "title": "Page module changed"}) + for number in (1, 2, 3)) + '\n', + encoding="utf-8", + ) + index.load() + with self.assertRaisesRegex(ValueError, "stale search cursor"): + index.search("module", limit=1, cursor=first["next_cursor"]) + with self.assertRaisesRegex(ValueError, "invalid search cursor"): + index.search("module", limit=1, cursor="vs1.invalid.1") + def test_generated_code_aliases_are_top_ranked(self): layout_results = self.index.search("ыев437", limit=3)["results"] bare_number_results = self.index.search("#437", limit=3)["results"] diff --git a/tests/test_v8std_mcp_runtime.py b/tests/test_v8std_mcp_runtime.py index 033e717..e462928 100644 --- a/tests/test_v8std_mcp_runtime.py +++ b/tests/test_v8std_mcp_runtime.py @@ -95,6 +95,7 @@ def test_validation_before_readiness_for_all_data_boundaries(self): with tempfile.TemporaryDirectory() as directory: facade = runtime.SnapshotIndex(site_url=LOCAL, cache_dir=Path(directory)) for call in (lambda: facade.search("x" * 501), lambda: facade.search("x", mode="invalid"), + lambda: facade.search("x", cursor="invalid"), lambda: facade.page("x" * 1001), lambda: facade.related("std437", relations=["bad"]), lambda: facade.explain_snippet("x" * 4001), lambda: facade.explain_diagnostics([1])): diff --git a/tests/test_v8std_mcp_tools_only.py b/tests/test_v8std_mcp_tools_only.py index 8b47f59..2486ce5 100644 --- a/tests/test_v8std_mcp_tools_only.py +++ b/tests/test_v8std_mcp_tools_only.py @@ -37,7 +37,7 @@ {"id_or_alias_or_url": "std437"}, {"snippet": "std437"}, {"codes": ["missing"]}) -# Hand-written expectations for the five inherited signatures, including defaults. +# Hand-written expectations for the five public signatures, including defaults. SCHEMAS = ( ("search", ["query"], { "query": {"title": "Query", "type": "string"}, @@ -45,6 +45,8 @@ "types": {"anyOf": [{"items": {"type": "string"}, "type": "array"}, {"type": "null"}], "default": None, "title": "Types"}, "mode": {"default": "hybrid", "title": "Mode", "type": "string"}, + "cursor": {"anyOf": [{"type": "string"}, {"type": "null"}], + "default": None, "title": "Cursor"}, }), ("page", ["id_or_alias_or_url"], { "id_or_alias_or_url": {"title": "Id Or Alias Or Url", "type": "string"}, @@ -144,6 +146,25 @@ def call_tool(test, rpc, name, arguments): class ToolsOnlyWireTests(unittest.TestCase): + def test_public_search_continues_past_the_first_fifty_results(self): + index = V8StdIndex( + pages_path=Path(__file__).resolve().parents[1] / "docs/ai/pages.jsonl", + vectors_path=Path(__file__).resolve().parents[1] / "docs/ai/search-vectors.jsonl", + ) + index.load() + with http_rpc(index) as rpc: + initialize(rpc) + assert_tool_catalog(self, rpc) + first = call_tool(self, rpc, "v8std_search", {"query": "модуль", "limit": 50}) + second = call_tool(self, rpc, "v8std_search", { + "query": "модуль", "limit": 50, "cursor": first["next_cursor"], + }) + self.assertGreater(first["total"], 50) + self.assertEqual(len(first["results"]), 50) + self.assertEqual(len(second["results"]), 50) + self.assertFalse({item["id"] for item in first["results"]} & + {item["id"] for item in second["results"]}) + def assert_ready_tools(self, rpc, site_url): results = [call_tool(self, rpc, name, arguments) for name, arguments in zip(TOOL_NAMES, TOOL_ARGUMENTS)] From 461f2cb626a6e2114357a949d8c1c478c69f9e59 Mon Sep 17 00:00:00 2001 From: Igor Apresov Date: Thu, 24 Sep 2026 23:50:56 +0300 Subject: [PATCH 2/4] docs(mcp): specify search cursor contract --- spec/mcp-surface-contract.md | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/spec/mcp-surface-contract.md b/spec/mcp-surface-contract.md index ddd84fe..c58b0a8 100644 --- a/spec/mcp-surface-contract.md +++ b/spec/mcp-surface-contract.md @@ -113,11 +113,19 @@ Markdown-ссылки в тексте. URL внутри примеров код | `limit` | integer | Нет | 10, ограничение 1–50 | | `types` | string[] или null | Нет | `standard`, `diagnostic`, `fix`, `pattern`, `service`; по умолчанию null | | `mode` | string | Нет | `hybrid` (по умолчанию), `exact`, `bm25`, `semantic` | +| `cursor` | string или null | Нет | Непрозрачный `next_cursor` предыдущей страницы; повторять с теми же `query`, `limit`, `types` и `mode` | Ответ: `query`, `normalized_query`, `mode` — строки; -`types` — отсортированный фильтр либо null; `results` — массив SearchEntry. +`types` — отсортированный фильтр либо null; `results` — массив SearchEntry +текущей страницы; `total` — точное число найденных записей; +`next_cursor` — непрозрачное продолжение либо null после последней страницы. +Ранжирование всей выдачи и отпечаток результатов связывают курсор с исходным +запросом и состоянием индекса. Повтор с тем же курсором возвращает ту же +страницу, а изменение ранжирования даёт ошибку устаревшего курсора вместо +пропуска или дублирования результатов. Неверный курсор также даёт ошибку. `semantic_enabled: boolean` присутствует при непустом нормализованном запросе; -для пустого нормализованного запроса это поле отсутствует, а `results = []`. +для пустого нормализованного запроса это поле отсутствует, а `results = []`, +`total = 0`, `next_cursor = null`. `hybrid` объединяет совпадения, текстовый и векторный поиск и связанные материалы. `exact` включает также варианты кодов и нечёткий поиск кодов: название режима From 732103488e73186d0edd9bdd7dbc2645213e1c6e Mon Sep 17 00:00:00 2001 From: Igor Apresov Date: Fri, 25 Sep 2026 12:51:31 +0300 Subject: [PATCH 3/4] perf(mcp): fingerprint ranked ids without projecting every hit --- runtime/v8std_mcp_index.py | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/runtime/v8std_mcp_index.py b/runtime/v8std_mcp_index.py index 622ffdf..ec58ec1 100644 --- a/runtime/v8std_mcp_index.py +++ b/runtime/v8std_mcp_index.py @@ -563,11 +563,13 @@ def search( [query, mode, sorted(allowed_types) if allowed_types else None, requested_limit], ensure_ascii=False, separators=(",", ":"), ).encode("utf-8")) - for score, page, candidate in entries: + # The index digests cover projected hit content. Hash only the + # ranked identity and score here: projecting every hit for a + # one-page request made ordinary searches pay for all pages. + for score, page, _candidate in entries: result_fingerprint.update(b"\0") result_fingerprint.update(json.dumps( - self._search_entry(page, score, candidate), - ensure_ascii=False, sort_keys=True, separators=(",", ":"), + [page["id"], score], ensure_ascii=False, separators=(",", ":"), ).encode("utf-8")) fingerprint = result_fingerprint.hexdigest() offset = 0 From f997aef0ebfb0349c836806bdafa7531eb18702f Mon Sep 17 00:00:00 2001 From: Igor Apresov Date: Fri, 25 Sep 2026 13:11:25 +0300 Subject: [PATCH 4/4] Bind search cursor to retrieval rules --- runtime/v8std_mcp_index.py | 10 +++++++++- tests/test_v8std_mcp_index.py | 30 ++++++++++++++++++++++++++++++ 2 files changed, 39 insertions(+), 1 deletion(-) diff --git a/runtime/v8std_mcp_index.py b/runtime/v8std_mcp_index.py index ec58ec1..a37c375 100644 --- a/runtime/v8std_mcp_index.py +++ b/runtime/v8std_mcp_index.py @@ -9,7 +9,7 @@ import threading import time from collections import Counter, defaultdict -from dataclasses import dataclass +from dataclasses import asdict, dataclass from difflib import SequenceMatcher from pathlib import Path from typing import Any @@ -394,6 +394,12 @@ def __init__( self.refresh_seconds = refresh_seconds self.request_timeout = request_timeout self.rules = RetrievalRules.load(rules_path) + # Retrieval rules can change projected hits (reasons and relations) + # without changing corpus bytes, vectors, or ranked scores. + self._rules_sha256 = hashlib.sha256(json.dumps( + [asdict(rule) for rule in self.rules.rules], + ensure_ascii=False, sort_keys=True, separators=(",", ":"), + ).encode("utf-8")).hexdigest() self._lock = threading.RLock() self._pages: list[dict[str, Any]] = [] self._pages_by_id: dict[str, dict[str, Any]] = {} @@ -559,6 +565,8 @@ def search( self._vector_metadata.sha256.encode("ascii") if self._vector_metadata else b"" ) result_fingerprint.update(b"\0") + result_fingerprint.update(self._rules_sha256.encode("ascii")) + result_fingerprint.update(b"\0") result_fingerprint.update(json.dumps( [query, mode, sorted(allowed_types) if allowed_types else None, requested_limit], ensure_ascii=False, separators=(",", ":"), diff --git a/tests/test_v8std_mcp_index.py b/tests/test_v8std_mcp_index.py index 6c217c3..5af3a89 100644 --- a/tests/test_v8std_mcp_index.py +++ b/tests/test_v8std_mcp_index.py @@ -109,6 +109,36 @@ def test_search_cursor_rejects_a_changed_index(self): with self.assertRaisesRegex(ValueError, "invalid search cursor"): index.search("module", limit=1, cursor="vs1.invalid.1") + def test_search_cursor_rejects_changed_rules_with_same_ranked_hits(self): + with tempfile.TemporaryDirectory() as temp_dir: + pages_path = Path(temp_dir) / "pages.jsonl" + rules_path = Path(temp_dir) / "retrieval-rules.yml" + pages_path.write_text( + '\n'.join(json.dumps({"id": f"std{number}", "type": "standard", "title": "Page module"}) + for number in (1, 2, 3)) + '\n', + encoding="utf-8", + ) + + def load_with_standard(standard_id): + rules_path.write_text( + f"rules:\n - id: example\n primary: std1\n standards: [{standard_id}]\n", + encoding="utf-8", + ) + index = V8StdIndex(pages_path=pages_path, rules_path=rules_path) + index.load() + return index + + original = load_with_standard("std2") + first = original.search("module", mode="bm25", limit=1) + changed = load_with_standard("std3") + changed_first = changed.search("module", mode="bm25", limit=1) + self.assertEqual(first["results"][0]["id"], changed_first["results"][0]["id"]) + self.assertEqual(first["results"][0]["score"], changed_first["results"][0]["score"]) + self.assertNotEqual(first["results"][0]["related_preview"], + changed_first["results"][0]["related_preview"]) + with self.assertRaisesRegex(ValueError, "stale search cursor"): + changed.search("module", mode="bm25", limit=1, cursor=first["next_cursor"]) + def test_generated_code_aliases_are_top_ranked(self): layout_results = self.index.search("ыев437", limit=3)["results"] bare_number_results = self.index.search("#437", limit=3)["results"]