Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
115 changes: 87 additions & 28 deletions runtime/v8std_mcp_index.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@
import threading
import time
from collections import Counter, defaultdict
from dataclasses import dataclass
from dataclasses import asdict, dataclass
from difflib import SequenceMatcher
from pathlib import Path
from typing import Any
Expand All @@ -36,6 +36,7 @@
MAX_ID_OR_ALIAS_CHARS = 1000
MAX_LIMIT = 50
DEFAULT_LIMIT = 10
SEARCH_CURSOR_RE = re.compile(r"^vs1\.([0-9a-f]{64})\.([1-9][0-9]*)$")
MAX_BODY_CHARS = 12000
MAX_BODY_LIMIT_CHARS = 30000
MAX_SNIPPET_CHARS = 4000
Expand Down Expand Up @@ -393,6 +394,12 @@ def __init__(
self.refresh_seconds = refresh_seconds
self.request_timeout = request_timeout
self.rules = RetrievalRules.load(rules_path)
# Retrieval rules can change projected hits (reasons and relations)
# without changing corpus bytes, vectors, or ranked scores.
self._rules_sha256 = hashlib.sha256(json.dumps(
[asdict(rule) for rule in self.rules.rules],
ensure_ascii=False, sort_keys=True, separators=(",", ":"),
).encode("utf-8")).hexdigest()
self._lock = threading.RLock()
self._pages: list[dict[str, Any]] = []
self._pages_by_id: dict[str, dict[str, Any]] = {}
Expand Down Expand Up @@ -491,42 +498,50 @@ def search(
types: list[str] | None = None,
mode: str = "hybrid",
limit: int | None = None,
cursor: str | None = None,
) -> dict[str, Any]:
query = require_text(query, "query", MAX_QUERY_CHARS)
requested_limit = clamp_limit(limit)
allowed_types = self._validate_types(types)
mode = self._validate_mode(mode)
cursor_match = self.validate_search_cursor(cursor)
self.refresh_if_needed()
normalized_query = normalize_query(query)
if not normalized_query:
if cursor_match is not None:
raise ValueError("invalid search cursor")
return {
"query": query,
"normalized_query": normalized_query,
"mode": mode,
"types": sorted(allowed_types) if allowed_types else None,
"results": [],
"total": 0,
"next_cursor": None,
}

query_tokens = self._query_tokens(query)
candidates: dict[str, dict[str, Any]] = {}
# A refresh must not replace the index between ranking, fingerprinting,
# and projection of one page. The lock is reentrant for lookup helpers.
with self._lock:
query_tokens = self._query_tokens(query)
candidates: dict[str, dict[str, Any]] = {}

if mode in {"hybrid", "exact"}:
self._add_exact_scores(candidates, query, normalized_query)
self._add_code_lookup_scores(candidates, query)
self._add_fuzzy_code_scores(candidates, query)
if mode in {"hybrid", "exact"}:
self._add_exact_scores(candidates, query, normalized_query)
self._add_code_lookup_scores(candidates, query)
self._add_fuzzy_code_scores(candidates, query)

if mode in {"hybrid", "bm25"}:
self._add_bm25_scores(candidates, query_tokens)
self._add_metadata_coverage_scores(candidates, query)
if mode in {"hybrid", "bm25"}:
self._add_bm25_scores(candidates, query_tokens)
self._add_metadata_coverage_scores(candidates, query)

if mode in {"hybrid", "semantic"}:
self._add_semantic_scores(candidates, query)
if mode in {"hybrid", "semantic"}:
self._add_semantic_scores(candidates, query)

if mode == "hybrid":
self._add_related_boosts(candidates)
if mode == "hybrid":
self._add_related_boosts(candidates)

entries = []
with self._lock:
entries = []
for page_id, candidate in candidates.items():
page = self._pages_by_id.get(page_id)
if not page:
Expand All @@ -541,18 +556,62 @@ def search(
continue
entries.append((score, page, candidate))

entries.sort(key=lambda item: (-item[0], concrete_rank(item[1]), item[1]["type"], item[1]["id"]))
return {
"query": query,
"normalized_query": normalized_query,
"mode": mode,
"types": sorted(allowed_types) if allowed_types else None,
"semantic_enabled": self._vector_metadata is not None and bool(self._vectors),
"results": [
self._search_entry(page, score, candidate)
for score, page, candidate in entries[:requested_limit]
],
}
entries.sort(key=lambda item: (-item[0], concrete_rank(item[1]), item[1]["type"], item[1]["id"]))
result_fingerprint = hashlib.sha256()
result_fingerprint.update(b"v8std-search-v1\0")
result_fingerprint.update(self._metadata.sha256.encode("ascii") if self._metadata else b"")
result_fingerprint.update(b"\0")
result_fingerprint.update(
self._vector_metadata.sha256.encode("ascii") if self._vector_metadata else b""
)
result_fingerprint.update(b"\0")
result_fingerprint.update(self._rules_sha256.encode("ascii"))
result_fingerprint.update(b"\0")
result_fingerprint.update(json.dumps(
[query, mode, sorted(allowed_types) if allowed_types else None, requested_limit],
ensure_ascii=False, separators=(",", ":"),
).encode("utf-8"))
# The index digests cover projected hit content. Hash only the
# ranked identity and score here: projecting every hit for a
# one-page request made ordinary searches pay for all pages.
for score, page, _candidate in entries:
result_fingerprint.update(b"\0")
result_fingerprint.update(json.dumps(
[page["id"], score], ensure_ascii=False, separators=(",", ":"),
).encode("utf-8"))
fingerprint = result_fingerprint.hexdigest()
offset = 0
if cursor_match is not None:
if cursor_match.group(1) != fingerprint:
raise ValueError("stale search cursor")
offset = int(cursor_match.group(2))
if offset >= len(entries):
raise ValueError("invalid search cursor")
next_offset = min(offset + requested_limit, len(entries))
return {
"query": query,
"normalized_query": normalized_query,
"mode": mode,
"types": sorted(allowed_types) if allowed_types else None,
"semantic_enabled": self._vector_metadata is not None and bool(self._vectors),
"results": [
self._search_entry(page, score, candidate)
for score, page, candidate in entries[offset:next_offset]
],
"total": len(entries),
"next_cursor": f"vs1.{fingerprint}.{next_offset}" if next_offset < len(entries) else None,
}

@staticmethod
def validate_search_cursor(cursor: str | None) -> re.Match[str] | None:
if cursor is None:
return None
if not isinstance(cursor, str) or len(cursor) > 128:
raise ValueError("invalid search cursor")
match = SEARCH_CURSOR_RE.fullmatch(cursor)
if match is None:
raise ValueError("invalid search cursor")
return match

def page(self, id_or_alias_or_url: str, *, body_limit: int = MAX_BODY_CHARS) -> dict[str, Any]:
id_or_alias_or_url = require_text(
Expand Down
7 changes: 5 additions & 2 deletions runtime/v8std_mcp_runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -83,13 +83,16 @@ def _present(self, generation, result):
def _lookup(self, generation, value):
return LinkCatalog(generation.canonical_site_url, self.site_url, generation.page_paths).lookup(value)

def search(self, query, *, types=None, mode="hybrid", limit=None):
def search(self, query, *, types=None, mode="hybrid", limit=None, cursor=None):
require_text(query, "query", MAX_QUERY_CHARS)
clamp_limit(limit)
V8StdIndex._validate_types(types)
V8StdIndex._validate_mode(mode)
V8StdIndex.validate_search_cursor(cursor)
generation = self._current()
return self._present(generation, generation.index.search(query, types=types, mode=mode, limit=limit))
return self._present(generation, generation.index.search(
query, types=types, mode=mode, limit=limit, cursor=cursor,
))

def page(self, id_or_alias_or_url, *, body_limit=MAX_BODY_CHARS):
require_text(id_or_alias_or_url, "id_or_alias_or_url", MAX_ID_OR_ALIAS_CHARS)
Expand Down
7 changes: 5 additions & 2 deletions runtime/v8std_mcp_server.py
Original file line number Diff line number Diff line change
Expand Up @@ -581,7 +581,9 @@ def build_server(
'identifier. Returns ranked page IDs, titles, descriptions, URLs, scores and match '
'reasons; no full article text. Known diagnostic codes are handled by '
'v8std_explain_diagnostics; BSL/SDBL source fragments by v8std_explain_snippet. An exact '
'page ID or URL can be read with v8std_get_page. An empty result means no match in this '
'page ID or URL can be read with v8std_get_page. Pass next_cursor back with the same '
'query, limit, types and mode to continue; an updated ranking rejects the old cursor. '
'An empty result means no match in this '
'corpus, not that the code is correct. Scores rank candidates and are not probabilities.'
),
)
Expand All @@ -590,8 +592,9 @@ def search(
limit: Annotated[int, Field(description="Maximum results, default 10; clamped to 1–50.")] = 10,
types: Annotated[list[str] | None, Field(description="Page types: standard, diagnostic, fix, pattern, service. null or [] means all types.")] = None,
mode: Annotated[str, Field(description="hybrid: combined search (default); exact: includes identifier variants and fuzzy code matches; bm25: text/metadata; semantic: indexed vectors.")] = "hybrid",
cursor: Annotated[str | None, Field(description="Opaque next_cursor from a previous search with identical query, limit, types and mode; rejects changed rankings.")] = None,
) -> dict[str, Any]:
result = index.search(query, types=types, mode=mode, limit=limit)
result = index.search(query, types=types, mode=mode, limit=limit, cursor=cursor)
tool_usage.record_search(query, result, system=current_client_system())
return result

Expand Down
12 changes: 10 additions & 2 deletions spec/mcp-surface-contract.md
Original file line number Diff line number Diff line change
Expand Up @@ -113,11 +113,19 @@ Markdown-ссылки в тексте. URL внутри примеров код
| `limit` | integer | Нет | 10, ограничение 1–50 |
| `types` | string[] или null | Нет | `standard`, `diagnostic`, `fix`, `pattern`, `service`; по умолчанию null |
| `mode` | string | Нет | `hybrid` (по умолчанию), `exact`, `bm25`, `semantic` |
| `cursor` | string или null | Нет | Непрозрачный `next_cursor` предыдущей страницы; повторять с теми же `query`, `limit`, `types` и `mode` |

Ответ: `query`, `normalized_query`, `mode` — строки;
`types` — отсортированный фильтр либо null; `results` — массив SearchEntry.
`types` — отсортированный фильтр либо null; `results` — массив SearchEntry
текущей страницы; `total` — точное число найденных записей;
`next_cursor` — непрозрачное продолжение либо null после последней страницы.
Ранжирование всей выдачи и отпечаток результатов связывают курсор с исходным
запросом и состоянием индекса. Повтор с тем же курсором возвращает ту же
страницу, а изменение ранжирования даёт ошибку устаревшего курсора вместо
пропуска или дублирования результатов. Неверный курсор также даёт ошибку.
`semantic_enabled: boolean` присутствует при непустом нормализованном запросе;
для пустого нормализованного запроса это поле отсутствует, а `results = []`.
для пустого нормализованного запроса это поле отсутствует, а `results = []`,
`total = 0`, `next_cursor = null`.

`hybrid` объединяет совпадения, текстовый и векторный поиск и связанные материалы.
`exact` включает также варианты кодов и нечёткий поиск кодов: название режима
Expand Down
83 changes: 83 additions & 0 deletions tests/test_v8std_mcp_index.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,89 @@ def test_hybrid_search_finds_standards_and_diagnostics(self):
self.assertIn("match_reasons", std_results[0])
self.assertIn("score_details", std_results[0])

def test_search_cursor_reads_past_fifty_without_reordering_or_duplication(self):
first = self.index.search("модуль", limit=50)
self.assertGreater(first["total"], 50)
self.assertEqual(len(first["results"]), 50)
self.assertIsNotNone(first["next_cursor"])
second = self.index.search("модуль", limit=50, cursor=first["next_cursor"])
self.assertEqual(second, self.index.search("модуль", limit=50, cursor=first["next_cursor"]))
self.assertFalse({item["id"] for item in first["results"]} &
{item["id"] for item in second["results"]})
with self.assertRaisesRegex(ValueError, "stale search cursor"):
self.index.search("форма", limit=50, cursor=first["next_cursor"])
with self.assertRaisesRegex(ValueError, "stale search cursor"):
self.index.search("модуль", limit=20, cursor=first["next_cursor"])
with self.assertRaisesRegex(ValueError, "stale search cursor"):
self.index.search("модуль", limit=50, mode="bm25", cursor=first["next_cursor"])
with self.assertRaisesRegex(ValueError, "stale search cursor"):
self.index.search("модуль", limit=50, types=["standard"], cursor=first["next_cursor"])

seen = first["results"] + second["results"]
page = second
while page["next_cursor"] is not None:
page = self.index.search("модуль", limit=50, cursor=page["next_cursor"])
seen.extend(page["results"])
self.assertEqual(len(seen), first["total"])
self.assertEqual(len({item["id"] for item in seen}), first["total"])
self.assertLessEqual(len(page["results"]), 50)

def test_search_cursor_rejects_a_changed_index(self):
with tempfile.TemporaryDirectory() as temp_dir:
pages_path = Path(temp_dir) / "pages.jsonl"
pages_path.write_text(
'\n'.join(json.dumps({"id": f"std{number}", "type": "standard", "title": "Page module"})
for number in (1, 2, 3)) + '\n',
encoding="utf-8",
)
index = V8StdIndex(
pages_path=pages_path,
vectors_path=REPO_ROOT / "docs" / "ai" / "search-vectors.jsonl",
)
index.load()
first = index.search("module", limit=1)
self.assertIsNotNone(first["next_cursor"])
pages_path.write_text(
'\n'.join(json.dumps({"id": f"std{number}", "type": "standard", "title": "Page module changed"})
for number in (1, 2, 3)) + '\n',
encoding="utf-8",
)
index.load()
with self.assertRaisesRegex(ValueError, "stale search cursor"):
index.search("module", limit=1, cursor=first["next_cursor"])
with self.assertRaisesRegex(ValueError, "invalid search cursor"):
index.search("module", limit=1, cursor="vs1.invalid.1")

def test_search_cursor_rejects_changed_rules_with_same_ranked_hits(self):
with tempfile.TemporaryDirectory() as temp_dir:
pages_path = Path(temp_dir) / "pages.jsonl"
rules_path = Path(temp_dir) / "retrieval-rules.yml"
pages_path.write_text(
'\n'.join(json.dumps({"id": f"std{number}", "type": "standard", "title": "Page module"})
for number in (1, 2, 3)) + '\n',
encoding="utf-8",
)

def load_with_standard(standard_id):
rules_path.write_text(
f"rules:\n - id: example\n primary: std1\n standards: [{standard_id}]\n",
encoding="utf-8",
)
index = V8StdIndex(pages_path=pages_path, rules_path=rules_path)
index.load()
return index

original = load_with_standard("std2")
first = original.search("module", mode="bm25", limit=1)
changed = load_with_standard("std3")
changed_first = changed.search("module", mode="bm25", limit=1)
self.assertEqual(first["results"][0]["id"], changed_first["results"][0]["id"])
self.assertEqual(first["results"][0]["score"], changed_first["results"][0]["score"])
self.assertNotEqual(first["results"][0]["related_preview"],
changed_first["results"][0]["related_preview"])
with self.assertRaisesRegex(ValueError, "stale search cursor"):
changed.search("module", mode="bm25", limit=1, cursor=first["next_cursor"])

def test_generated_code_aliases_are_top_ranked(self):
layout_results = self.index.search("ыев437", limit=3)["results"]
bare_number_results = self.index.search("#437", limit=3)["results"]
Expand Down
1 change: 1 addition & 0 deletions tests/test_v8std_mcp_runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -95,6 +95,7 @@ def test_validation_before_readiness_for_all_data_boundaries(self):
with tempfile.TemporaryDirectory() as directory:
facade = runtime.SnapshotIndex(site_url=LOCAL, cache_dir=Path(directory))
for call in (lambda: facade.search("x" * 501), lambda: facade.search("x", mode="invalid"),
lambda: facade.search("x", cursor="invalid"),
lambda: facade.page("x" * 1001), lambda: facade.related("std437", relations=["bad"]),
lambda: facade.explain_snippet("x" * 4001),
lambda: facade.explain_diagnostics([1])):
Expand Down
Loading