diff --git a/core/ingest/code_corpus.py b/core/ingest/code_corpus.py index b5cee695..224b2e13 100644 --- a/core/ingest/code_corpus.py +++ b/core/ingest/code_corpus.py @@ -25,11 +25,20 @@ source order, windowed as CANONICAL (header-free) prose and prefixed for retrieval by a single `# {path}` line — it lives in the note neighbourhood. -The three are joined by line-range coordinates carried ON the rows (the §2.4 L2a fiber — no edge -rows). `digest` = the git blob sha (git is already the content-addressed raw store for code), so -group-by-digest gives file = source object, chunks = members, UNCHANGED. Derivation is a PURE -function of (path, source): re-running yields bit-identical chunks (F-CI2 re-derivability). All -embedding is LOCAL (the core embedder) — zero network egress (non-negotiable #1). +Derivation is a PURE function of (path, source): re-running yields bit-identical chunks (F-CI2 +re-derivability). All embedding is LOCAL (the core embedder) — zero network egress +(non-negotiable #1). + +[banner: correction] The three projections WERE joined by line-range coordinates carried ON the +vector rows, and `digest` (the git blob sha) made group-by-digest yield "file = source object, +chunks = members". **Neither is true of a code row any more** (dn-vector-membership-store D1, +bp-152). The vector plane holds ONE row per distinct idea-atom `(layer, content_hash)`, +corpus-wide and append-only; ALL occupancy — which `(path, blob_sha)` version holds which atom, at +which slot and lines, and whether that occupancy is current — moved into the membership relation +(`core/stores/memberships.py`). A version is a FIBER `M(path, blob_sha)`, the source object is +that fiber, and a code consumer resolves a hit through the membership join (D3), never through +group-by-digest. The measured payoff is the reason: 52,755 duplicated embeds over the full ledger +history become 22,502 atoms (2.34×, D7), and a revert or a `git mv` costs zero geometry. [banner: correction] A chunk's IDENTITY (`content_hash`) hashed its full embed text — coordinate header included — so a `git mv` re-hashed every chunk of the file and every `(path, slot)` lineage @@ -55,14 +64,21 @@ from __future__ import annotations from collections.abc import Sequence -from dataclasses import dataclass +from dataclasses import dataclass, field from hashlib import sha256 from pathlib import Path from typing import Any, Protocol from core.kernel.ingest.chunk import chunk_text from core.kernel.provenance import Provenance +from core.stores.memberships import ( + CurrencyReport, + EmbedderIdentity, + Membership, + MembershipStore, +) from core.stores.vectorstore import ( + ATOM_ROW_SHED, LAYER_CODE_AST, LAYER_CODE_TEXT, LAYER_CODEDOC, @@ -83,18 +99,33 @@ def embed_documents(self, texts: list[str]) -> list[list[float]]: ... @dataclass(frozen=True) class CodeChunk: """One embeddable code chunk with its fiber coordinates. `layer` discriminates the projection; - `(qualname, line_start, line_end)` are the §2.4 backpointers carried on the row. + `(qualname, slot_line_start, slot_line_end)` are the §2.4 backpointers, and they travel to the + MEMBERSHIP row now (bp-152 D1), not to the vector row. TWO renderings, deliberately different (D0): `text` is the EMBED rendering and KEEPS its coordinate header (retrieval context, R7); `canonical_body` is the IDENTITY input and is header-free. Every chunker passes the canonical body from the site that already holds it — it is NEVER re-derived by re-parsing `text`, so a body line that legitimately begins with `#` can - never be mistaken for a coordinate header (a wrong strip is silent identity corruption).""" + never be mistaken for a coordinate header (a wrong strip is silent identity corruption). + + [banner: correction] `line_start` / `line_end` are RENAMED to `slot_line_start` / + `slot_line_end` (dn-vector-membership-store Amendment A2, owner-ruled 2026-08-06 on issue #34). + The stored values do not change; the name does, because the old name licensed a wrong reading. + **They are the SLOT's declared extent — where the symbol lives — never the atom's text + coverage.** L0a partitions by INNERMOST OWNER, so a class's chunk holds the class statement, + its docstring and its attributes but NOT its methods (which became their own chunks) — while + the emitted coordinates are `owner.lineno, owner.end_lineno`, the owner's full declared span. + They coincide exactly for leaf symbols, which is why every leaf-symbol fixture is blind to the + divergence; the module shell is the maximal case, carrying `1..n` (the ENTIRE file) for a few + lines of preamble. The values are also non-contiguous in general: a symbol with nested children + owns lines scattered across its span. A consumer that wants "where is this symbol" reads the + span; a consumer that wants the atom's content reads `text`. This is the intended behavior, not + a defect to fix — the fix was to stop letting the field name hide the difference.""" layer: str qualname: str - line_start: int - line_end: int + slot_line_start: int + slot_line_end: int text: str # the embed rendering: header + body (headerless for L0b) canonical_body: str # the identity input: header-free (== text for L0b) @@ -125,7 +156,7 @@ def _l0a_chunks(path: str, lines: list[str], shape: FileShape, *, n = len(lines) # group source-line numbers by innermost owner (qualname, or '' for the module shell) owned: dict[str, list[int]] = {} - coords: dict[str, tuple[str, int, int]] = {} # qualname -> (header, line_start, line_end) + coords: dict[str, tuple[str, int, int]] = {} # qualname -> (header, slot extent start/end) for i in range(1, n + 1): owner = _innermost_owner(shape.symbols, i) if owner is None: @@ -161,10 +192,10 @@ def _l0a_chunks(path: str, lines: list[str], shape: FileShape, *, # ── L0b: the windowed textual reading — chunk_text over raw source ─────────────────────── def _locate_span(chunk_body: str, lines: list[str], cursor: int) -> tuple[int, int, int]: - """Best-effort (line_start, line_end, next_cursor) for an L0b window — located by matching the - window's first/last non-blank line back into the source (windows overlap by design, so the - cursor only hints, never hard-bounds). (0, 0) when unlocatable. L0b coords feed only the - [INFERENCE]-graded M-C8 join, so best-effort is the right cost here.""" + """Best-effort (slot_line_start, slot_line_end, next_cursor) for an L0b window — located by + matching the window's first/last non-blank line back into the source (windows overlap by + design, so the cursor only hints, never hard-bounds). (0, 0) when unlocatable. L0b coords feed + only the [INFERENCE]-graded M-C8 join, so best-effort is the right cost here.""" body = [ln.strip() for ln in chunk_body.split("\n") if ln.strip()] if not body: return (0, 0, cursor) @@ -243,58 +274,225 @@ def derive_code_chunks(path: str, source: str, *, # ── the structural CODE mint — row assembly with NO provenance parameter (F-CI1) ───────── -def code_rows(path: str, blob_sha: str, chunks: Sequence[CodeChunk], - vectors: Sequence[list[float]], *, current: bool = True) -> list[dict[str, Any]]: - """Assemble vector-store rows for one file version. Provenance is HARDCODED `CODE` — there is - NO parameter, so a caller physically cannot launder code into an authored class (F-CI1). `id` - is `(source_path, layer, chunk_hash)` — doc+layer-scoped and content-addressed, so an unchanged - chunk keeps its point across versions and two layers with identical text stay distinct. `digest` - is the git blob sha, so group-by-digest yields file = source object, its chunks = members. - As of D0 the hash is CANONICAL-BODY-scoped (header-free), so a renamed file's chunks keep their - point too — the id SHAPE still carries the path; the path-free `(layer, hash)` atom id is D1's - change (bp-152), not this one. - - `current` (dn-temporal-code-corpus D2, bp-099) marks whether this version is the path's HEAD - projection: the sync lands the new version `current=True` and flips the superseded one to False - (keep-and-link — never a delete); a history backfill lands each version `current = (blob is - HEAD's blob)`. `current` is a KEEP-AND-LINK flag, not a provenance parameter (F-CI1 intact).""" +def atom_id(chunk: CodeChunk) -> str: + """The atom's identity (D1): `"{layer}:{content_hash}"` — PATH-FREE and corpus-wide. + + The one place the id shape is spelled, so the vector row and the membership row can never + disagree about what an atom is. Dropping the path from the id is the whole atom model in one + character-level change: it promotes `code_rows`' old per-path dedup to corpus-wide dedup, so + the same body in two files is ONE point with two memberships (PD-1, owner-ruled in). The layer + stays inside identity because two layers with identical text are different readings, not the + same idea (the D1 stratum fence, carried as a test invariant).""" + return f"{chunk.layer}:{chunk.content_hash}" + + +def code_rows(chunks: Sequence[CodeChunk], vectors: Sequence[list[float]], *, + current: bool = False) -> list[dict[str, Any]]: + """Assemble ATOM rows — one per distinct `(layer, content_hash)` (D1). Provenance is HARDCODED + `CODE`: there is NO parameter, so a caller physically cannot launder code into an authored class + (F-CI1). + + [banner: correction] The old docstring said `id` is `(source_path, layer, chunk_hash)` — + "doc+layer-scoped" — and that `digest` is the git blob sha "so group-by-digest yields file = + source object, its chunks = members". **Both clauses stop being true here.** Under D1 the id is + `(layer, content_hash)`, corpus-wide; the source object is now the MEMBERSHIP FIBER + `M(path, blob_sha)` (`code_memberships`, below), and group-by-digest is not the code lane's + path at all — `VectorStore.all_rows` structurally keeps shed atom rows out of it, because a + grouping keyed on a column these rows do not have produces one bogus set rather than an error + (bp-152 Item 3; the note's §3 Q5). + + **The shed, stated exactly.** The occupancy columns — `source_path`, `digest`, `title`, + `chunk_index`, `qualname`, `line_*` — leave the ROW, not the schema: note rows still carry + them and no prose-lane consumer changes. (`title` is shed with them because on a code row it + WAS the path under another name; leaving it would stamp each shared atom with its first-landed + path — a coordinate that lies for every other occupancy, which is exactly what D0's consequence + note forbids relying on. Occupancy resolves from memberships, never from the row.) + **`provenance` STAYS**, and that is load-bearing rather than incidental: the mirror firewall is + a row PREFILTER — `provenance IN (...)` with `prefilter=True` in `VectorStore.search` — so + shedding the column would not weaken the firewall, it would REMOVE it, silently and with no + failing call anywhere. + + **The dedup here is the atom side and ONLY the atom side.** `by_id.setdefault` collapses + duplicates, which is correct for geometry — two identical bodies are one idea — and WRONG for + occupancy: two byte-identical L0b windows in one blob are TWO memberships with distinct + `chunk_index` (the F5 multiset pin). `code_memberships` therefore builds its own list and never + reuses this dict. + + `current` is the `current_any` reading now (D1): does ANY current membership contain this atom? + A freshly landed atom defaults to **False** — it has no occupancy yet at insert time, since D8 + puts the vector insert BEFORE the fiber write — and the lander raises it in step 5 for exactly + the atoms whose current-membership count crossed 0→1.""" by_id: dict[str, dict[str, Any]] = {} - for idx, (ch, vec) in enumerate(zip(chunks, vectors, strict=True)): - rid = f"{path}:{ch.layer}:{ch.content_hash}" + for ch, vec in zip(chunks, vectors, strict=True): + rid = atom_id(ch) row: dict[str, Any] = { + **ATOM_ROW_SHED, # occupancy lives in memberships now (D1) "id": rid, - "digest": blob_sha, - "title": path, - "source_path": path, - "chunk_index": idx, + "title": "", "provenance": Provenance.CODE.value, # ← hardcoded; no parameter anywhere above "text": ch.text, "layer": ch.layer, - "qualname": ch.qualname, - "line_start": ch.line_start, - "line_end": ch.line_end, "current": current, "vector": vec, } - by_id.setdefault(rid, row) # one point per (path, layer, content) + by_id.setdefault(rid, row) # one point per (layer, content) — corpus-wide return list(by_id.values()) +def code_memberships(path: str, blob_sha: str, + chunks: Sequence[CodeChunk]) -> list[Membership]: + """The version's FIBER: one membership row per chunk, in derivation order (D1/D2 step 3). + + This is the A2 translation point — `CodeChunk.slot_line_*` becomes `Membership.slot_line_*` + with the name intact, so the "declared extent, not text coverage" reading survives the trip to + storage instead of being re-lost at the boundary. + + **One row per CHUNK, never per distinct atom.** `chunk_index` is the position in + `derive_code_chunks`' output, which is a pure deterministic function of `(path, source)` + (F-CI2), so the key `(path, blob_sha, layer, chunk_index)` is stable and re-derivable — and two + identical windows in one blob keep both occupancies instead of colliding. Rows land + `current=False`; currency is not a property of the fiber's construction but of reconciliation + against the path's HEAD (D2 step 4), which is the only place that decides it.""" + return [ + Membership( + path=path, blob_sha=blob_sha, layer=ch.layer, chunk_index=idx, + content_id=atom_id(ch), + slot=ch.qualname, # L0a's symbol; '' for L0b/L1 (R4, no re-slot) + slot_line_start=ch.slot_line_start, slot_line_end=ch.slot_line_end, + current=False, tombstoned=False, + ) + for idx, ch in enumerate(chunks) + ] + + +# ── land(): the D2 write path — five steps, and step 4 is the one that must never be skipped ── + +@dataclass(frozen=True) +class LandReport: + """What one `land()` actually did. `atoms_embedded == 0` on a re-land is the reuse claim; the + CURRENCY numbers are what prove idempotence, because a do-nothing lander also embeds zero.""" + + atoms_embedded: int = 0 + atoms_reused: int = 0 + membership_rows: int = 0 # NEW occupancy rows; 0 on a re-land — the fiber already stands + currency: CurrencyReport = field(default_factory=CurrencyReport) + current_any_raised: int = 0 + current_any_lowered: int = 0 + + +@dataclass +class CodeLander: + """`land(path, blob_sha, chunks)` — the D2 write path, in the D8 order. + + Vector inserts FIRST (append-only; an unreferenced atom is dormant geometry, harmless), the + membership fiber SECOND (one SQLite transaction — the reference truth), currency reconciliation + and `current_any` maintenance LAST (both re-derivable, so a crash anywhere is repaired by the + next land or by `repair_current_any`). + + ⚑ **Re-landing is idempotent BECAUSE reconciliation converges, not because the call + short-circuits.** Step 4 runs even when step 3 wrote nothing. The tempting "the fiber already + exists, so return" is the C1 bug in its exact original form: on A → B → A the fiber for blob A + already exists carrying `current=false`, so a short-circuit leaves **B** marked HEAD — silent + corruption of every default (current-view) read, with nothing raised and nothing logged. The + repo already learned this once at note grain (`core/stores/versions.py:22-27`).""" + + vectors: VectorStore + memberships: MembershipStore + embedder: _Embedder + embedder_identity: EmbedderIdentity + + def land(self, path: str, blob_sha: str, chunks: Sequence[CodeChunk], *, + head_blob_sha: str | None = None) -> LandReport: + """Land one file version. `head_blob_sha` names the path's CURRENT HEAD blob — it defaults + to the version being landed (the incremental case) and is passed explicitly by a history + backfill, where the version being landed is usually NOT head.""" + head = blob_sha if head_blob_sha is None else head_blob_sha + + # (1) canonical identity per chunk — D0/bp-151: the hash is over the header-free body, so a + # rename mints nothing and the atom survives `git mv`. + ids = [atom_id(ch) for ch in chunks] + + # (2) insert only the atoms absent from the plane — the embed step; everything else is + # reuse by construction. Presence is keyed to (layer, content_hash) AND the embedder + # identity: a model change must invalidate every reuse, or two geometries share one ANN + # space and no downstream measurement can tell. + known = self.memberships.known_atoms(ids, self.embedder_identity) + fresh: dict[str, CodeChunk] = {} + for ch, cid in zip(chunks, ids, strict=True): + if cid not in known: + fresh.setdefault(cid, ch) # atoms dedup here; memberships never do + reused = len(set(ids)) - len(fresh) + if fresh: + vecs = self.embedder.embed_documents([c.text for c in fresh.values()]) + # current=False: at insert time the atom has no occupancy yet (D8 puts vectors first), + # so `current_any` is false until step 5 sees it cross 0→1. + self.vectors.add(code_rows(list(fresh.values()), vecs, current=False)) + self.memberships.record_atoms( + [(cid, ch.layer) for cid, ch in fresh.items()], self.embedder_identity) + + # The step-5 "before" reading, taken BEFORE the fiber write. The candidate set is the + # atoms this land could possibly move: the version's own atoms plus everything already + # occupying this path (reconciliation is path-scoped, so nothing else can cross). + candidates = set(ids) | self.memberships.atom_ids_of_path(path) + before = self.memberships.currently_held(candidates) + + # (3) write the fiber; an existing fiber's rows STAND (pure derivation ⇒ fiber equality) + written = self.memberships.write_fiber(code_memberships(path, blob_sha, chunks)) + + # (4) currency reconciliation — NEVER skipped, even when (3) was a no-op (the C1 case) + currency = self.memberships.reconcile_currency(path, head) + + # (5) maintain `current_any` on exactly the atoms whose current-membership count crossed + # 0↔1 — `current` is a lance column, so an unconditional flip would rewrite fragments + # for every atom of every landing (the §3 physical-maintenance pin). + after = self.memberships.currently_held(candidates) + raised = self.vectors.set_current_any(after - before, True) + lowered = self.vectors.set_current_any(before - after, False) + + return LandReport(atoms_embedded=len(fresh), atoms_reused=reused, + membership_rows=written, currency=currency, + current_any_raised=raised, current_any_lowered=lowered) + + def reconcile(self, path: str, head_blob_sha: str) -> LandReport: + """Steps 4–5 alone, for a path whose HEAD fiber already stands. + + This exists so the incremental sync can honor the C1 rule without re-deriving chunks for + every unchanged file on every pass. "Unchanged blob ⇒ skip the path entirely" is the same + short-circuit at one level up: after A → B → A the HEAD fiber exists, the file looks + unchanged, and B is left current forever. Reconciliation is two counting queries per path, + so convergence costs nothing worth trading for that.""" + candidates = self.memberships.atom_ids_of_path(path) + before = self.memberships.currently_held(candidates) + currency = self.memberships.reconcile_currency(path, head_blob_sha) + after = self.memberships.currently_held(candidates) + return LandReport(currency=currency, + current_any_raised=self.vectors.set_current_any(after - before, True), + current_any_lowered=self.vectors.set_current_any(before - after, False)) + + def supersede_path(self, path: str) -> LandReport: + """A vanished file: every fiber of `path` goes `current=false`, nothing is deleted + (keep-and-link, D2). Expressed as reconciliation against a blob no fiber has, so there is + ONE currency mechanism rather than a second, subtly different one.""" + return self.reconcile(path, "") + + # ── incremental sync + the seed — blob-sha-keyed, unchanged file = zero embeds ─────────── @dataclass class CodeSyncReport: - embedded_rows: int = 0 + embedded_rows: int = 0 # ATOM rows inserted (D1) — an unchanged/duplicate atom is 0 changed_files: int = 0 unchanged_files: int = 0 deleted_files: int = 0 - superseded_rows: int = 0 # rows flipped current=true→false, RETAINED (keep-and-link, D2) + superseded_rows: int = 0 # MEMBERSHIP rows flipped current=true→false, RETAINED (D2) parse_failures: int = 0 # blobs that failed AST-parse → L0b-only, still embedded (D1) + membership_rows: int = 0 # new occupancies recorded this pass def __str__(self) -> str: return (f"embedded_rows={self.embedded_rows} changed={self.changed_files} " f"unchanged={self.unchanged_files} deleted={self.deleted_files} " - f"superseded_rows={self.superseded_rows} parse_failures={self.parse_failures}") + f"superseded_rows={self.superseded_rows} parse_failures={self.parse_failures} " + f"membership_rows={self.membership_rows}") @dataclass @@ -311,43 +509,59 @@ class CodeCorpusSync: repo: Path store: VectorStore embedder: _Embedder + memberships: MembershipStore + embedder_identity: EmbedderIdentity max_chars: int = _DEFAULT_MAX_CHARS overlap_chars: int = _DEFAULT_OVERLAP_CHARS - def _embed_and_land(self, path: str, blob_sha: str, source: str, *, - current: bool = True) -> int: - """Derive → embed → land one file version's rows. Keep-and-link (D2): it NEVER deletes the - path's prior projection — the caller flips the superseded version to `current=false` first - (`store.supersede_source`). `current` marks whether this version is HEAD's projection.""" + @property + def lander(self) -> CodeLander: + return CodeLander(vectors=self.store, memberships=self.memberships, + embedder=self.embedder, embedder_identity=self.embedder_identity) + + def _land(self, path: str, blob_sha: str, source: str, *, + head_blob_sha: str | None = None) -> LandReport: + """Derive → land one file version through the D2 write path.""" chunks = derive_code_chunks(path, source, max_chars=self.max_chars, overlap_chars=self.overlap_chars) if not chunks: - return 0 - vectors = self.embedder.embed_documents([c.text for c in chunks]) - rows = code_rows(path, blob_sha, chunks, vectors, current=current) - return self.store.add(rows) + return LandReport() + return self.lander.land(path, blob_sha, chunks, head_blob_sha=head_blob_sha) def sync(self) -> CodeSyncReport: + """[banner: correction] The D-fiber state WAS the store's own set of CODE + `(source_path, digest)` pairs. Atom rows carry neither column (D1), so the state re-homes + to the membership store's `(path, blob_sha)` fibers — the same number, a sturdier home + (the note's §6 re-home (1), applied here; the daemon's incompleteness probe is bp-153's). + + A path whose HEAD fiber already stands is UNCHANGED and re-derives nothing — but it is + still reconciled. Skipping it outright is the C1 short-circuit one level up: after + A → B → A the HEAD fiber exists, the file reads as unchanged, and B stays current.""" report = CodeSyncReport() head = list_py_blobs(self.repo, "HEAD") # [(path, blob_sha)] - code_now = self.store.all_rows(provenances={Provenance.CODE}) - present_pd = {(str(r["source_path"]), str(r["digest"])) for r in code_now} - present_paths = {str(r["source_path"]) for r in code_now} + present_fibers = set(self.memberships.fibers()) + present_paths = {p for p, _ in present_fibers} + lander = self.lander - changed = [(p, b) for p, b in head if (p, b) not in present_pd] + changed = [(p, b) for p, b in head if (p, b) not in present_fibers] report.changed_files = len(changed) report.unchanged_files = len(head) - len(changed) deleted = present_paths - {p for p, _ in head} - for p in deleted: # vanished file: keep rows, flip current=false (D2) - report.superseded_rows += self.store.supersede_source(p) + for p in sorted(deleted): # vanished file: keep rows, flip current=false (D2) + report.superseded_rows += lander.supersede_path(p).currency.superseded report.deleted_files = len(deleted) + for path, blob_sha in head: # converge EVERY head version (the C1 rule) + if (path, blob_sha) in present_fibers: + report.superseded_rows += lander.reconcile(path, blob_sha).currency.superseded + blobs = read_py_blobs(self.repo, sorted({b for _, b in changed})) for path, blob_sha in changed: - report.superseded_rows += self.store.supersede_source(path) # keep the old version (D2) - report.embedded_rows += self._embed_and_land(path, blob_sha, blobs[blob_sha], - current=True) + landed = self._land(path, blob_sha, blobs[blob_sha]) + report.embedded_rows += landed.atoms_embedded + report.membership_rows += landed.membership_rows + report.superseded_rows += landed.currency.superseded return report def seed(self) -> CodeSyncReport: @@ -365,12 +579,17 @@ def backfill(self, versions: Sequence[tuple[str, str]]) -> CodeSyncReport: `current = (blob is that path's HEAD blob)`, so backfilling into an un-seeded store also marks HEAD correctly and every superseded version `current=false`. A parse-fail blob still embeds (L0b windows + module shell, `derive_code_chunks` degrades — never a hard stop) and - is counted. Store writes stay on the caller (the supervisor handler), single-writer kept.""" + is counted. Store writes stay on the caller (the supervisor handler), single-writer kept. + + [banner: correction] "Already in the store" is now "already has a FIBER" (D1 — the atom row + carries no `(source_path, digest)` to test). Idempotence is unchanged in kind and stronger + in fact: a re-run derives nothing for a version whose fiber stands, and a version whose + atoms are all already in the plane costs zero embeds even the FIRST time it is landed — + which is the whole point of the split (D7's 52,755 → 22,502 measured).""" report = CodeSyncReport() head = dict(list_py_blobs(self.repo, "HEAD")) # path -> HEAD blob_sha - code_now = self.store.all_rows(provenances={Provenance.CODE}) - present_pd = {(str(r["source_path"]), str(r["digest"])) for r in code_now} - todo = [(p, b) for (p, b) in dict.fromkeys(versions) if (p, b) not in present_pd] + present_fibers = set(self.memberships.fibers()) + todo = [(p, b) for (p, b) in dict.fromkeys(versions) if (p, b) not in present_fibers] if not todo: return report blobs = read_py_blobs(self.repo, sorted({b for _, b in todo})) @@ -380,21 +599,27 @@ def backfill(self, versions: Sequence[tuple[str, str]]) -> CodeSyncReport: continue if parse_source(path, blob_sha, source).parse_error: report.parse_failures += 1 - landed = self._embed_and_land(path, blob_sha, source, - current=head.get(path) == blob_sha) - if landed: - report.embedded_rows += landed + landed = self._land(path, blob_sha, source, + head_blob_sha=head.get(path, blob_sha)) + if landed.membership_rows: + report.embedded_rows += landed.atoms_embedded + report.membership_rows += landed.membership_rows + report.superseded_rows += landed.currency.superseded report.changed_files += 1 return report def build_code_corpus_sync(config: Any = None, *, repo: Path | None = None, embedder: _Embedder | None = None) -> CodeCorpusSync: - """Wire a CodeCorpusSync against the configured vector store + local embedder + repo root.""" + """Wire a CodeCorpusSync against the configured vector store, membership store, local embedder + and repo root. The membership store and the embedder identity are REQUIRED fields rather than + optional ones: a lander without an occupancy record is not a lander, and a reuse decision + without an embedder identity is the geometry-mixing bug (D2 step 2's pin) waiting to happen.""" import subprocess from core.ingest.embed import build_embedder from core.kernel.config import get_config + from core.stores.memberships import open_membership_store cfg = config or get_config() root = repo or Path(subprocess.run(["git", "rev-parse", "--show-toplevel"], @@ -403,4 +628,6 @@ def build_code_corpus_sync(config: Any = None, *, repo: Path | None = None, repo=root, store=VectorStore(cfg.paths.vector_store, dim=cfg.embedding.dim), embedder=embedder or build_embedder(cfg), + memberships=open_membership_store(cfg), + embedder_identity=EmbedderIdentity.from_config(cfg), ) diff --git a/core/stores/memberships.py b/core/stores/memberships.py new file mode 100644 index 00000000..38018fe1 --- /dev/null +++ b/core/stores/memberships.py @@ -0,0 +1,530 @@ +# ── Family 1 boundary (labelings & information-flow) · symbols in docs/NOTATION.md ── +# OBJECT: the membership relation M ⊆ V × O — which idea-atoms occupy which (path, blob_sha) +# versions, at which slot and lines, and whether that occupancy is current +# (dn-vector-membership-store D1/D2/D8, warrant finding-0168). +# INVARIANT: MULTISET, never a set — the key is the occupancy's coordinates +# (path, blob_sha, layer, chunk_index), so two byte-identical windows in one blob are +# TWO rows. No occupancy vanishes by key collision. `slot` is the qualname for L0a +# only; L0b/L1 are membership-only and honestly chainless (R4). +# ENFORCED: structural — this module lives in `core/stores/` (OUTER ring) and is NEVER imported +# by `core/kernel/**`. Memberships reach the kernel as DATA, through the existing +# `RowSource`-shaped protocol (`core/kernel/stores/sourceset.py`); a kernel import +# would mechanically demote `sourceset` from the inner-ring fixed point +# (`core/kernel/rings.py:36-41`, the C5/D3 pin). If the seam ever cannot stay +# data-shaped, the design note is explicit: that is a finding, not an import. +"""The membership store — occupancy as a first-class relation (dn-vector-membership-store, bp-152). + +A point is a point: geometry, assertion-free. **Meaning lives in membership** — who contains it — +and history in **lineage** — which occupancy chains pass through it. The vector plane +(`core/stores/vectorstore.py`) holds ONE row per distinct idea-atom `(layer, content_hash)`, +append-only; this store holds everything that used to be duplicated onto those rows once per +version: + + * **A version is a fiber.** `M(path, blob_sha)` — the complete projection of one file version. + Landing a version writes its fiber; re-landing the same blob writes nothing new, because + derivation is a pure function of `(path, source)` so the fiber is equal by construction. + * **A re-land is idempotent BECAUSE reconciliation converges** — never because the call + short-circuits. This is the C1 lesson the repo has already learned once + (`core/stores/versions.py:22-27`: content-keyed identity cannot hold a revert). On A → B → A + the fiber for blob A already EXISTS with `current=false`; a lander that returns early on + "fiber exists" leaves **B** current, which is silent D3 corruption with nothing to see. So + `reconcile_currency` runs on every land, including the ones that wrote no row. + * **`current_any` is a cache, this store is the truth** (D8/R3). The lance `current` column is + the cheap ANN prefilter; membership rows are what it caches, and `repair_current_any` rebuilds + it from them. Write order is vector inserts FIRST (append-only, an unreferenced atom is + harmless dormant geometry), fiber SECOND (one SQLite transaction), currency LAST — every step + after the first is re-derivable, so a crash anywhere is repaired by the next land. + * **Append-only, with ONE removal.** No API here deletes a vector except `purge_atom` (D5, + finding-0164 — owner-gated, privacy outranks lineage), and a purge leaves a RECORDED HOLE: the + row is gone and its memberships are tombstoned, never silently dropped. + +Lineage is DERIVED here, never stored (PD-3): `slot_runs` / `slot_edges` collapse a path's +first-parent blob chain into runs of equal occupants per slot, so a revert reads as 3 runs and 2 +edges rather than being flattened away. Chain members are a strict subset of the version set (D4/F3 +— a side-branch fiber is a real member of M with no place on any chain), so the edge invariants are +quantified over the chain that is handed in, never over all fibers. +""" + +from __future__ import annotations + +import sqlite3 +from collections.abc import Iterable, Sequence +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path + +from core.kernel.config import Config +from core.stores.vectorstore import LAYER_CODE_AST, VectorStore, is_code_atom_row + + +def _utcnow() -> str: + return datetime.now(UTC).replace(tzinfo=None).isoformat(timespec="seconds") + + +@dataclass(frozen=True) +class EmbedderIdentity: + """The geometry an atom's stored vector came from — `EmbeddingConfig.model` + `dim`. + + Embed REUSE (D2 step 2, "insert only the atoms absent from the plane") is only valid within one + embedder, and the owner pinned that explicitly (2026-08-01, issue #27 sub-confirmation 2): an + atom already in the plane carries a vector from whichever embedder landed it, and serving it + beside freshly-embedded atoms mixes two geometries in ONE ANN space — a corruption no + downstream measurement can detect. So atom PRESENCE is keyed to `(layer, content_hash)` **and** + this identity; a hit whose embedder differs from the live config is not a hit. + + `query_instruction` is deliberately excluded: it conditions *queries*, not stored document + vectors, so two configs differing only there share a geometry.""" + + model: str + dim: int + + @classmethod + def from_config(cls, config: Config) -> EmbedderIdentity: + return cls(model=config.embedding.model, dim=config.embedding.dim) + + +@dataclass(frozen=True) +class Membership: + """One occupancy: "atom `content_id` sits in version `(path, blob_sha)` at this slot". + + The key is `(path, blob_sha, layer, chunk_index)` — the occupancy's COORDINATES, not its + content. That is the multiset pin (the note's F5): two byte-identical L0b windows in one blob + differ only in `chunk_index`, and keying on content would silently merge them. + + ⚑ `slot_line_start` / `slot_line_end` are the **SLOT's declared extent** — where the symbol + lives — and **never the atom's text coverage** (Amendment A2, issue #34). They coincide for + leaf symbols. For a symbol with nested children the children's lines are carved OUT of the + parent's chunk while the span still reports the parent's full `lineno..end_lineno`; for the + module shell the span is the ENTIRE FILE by construction. The columns are named for what they + measure precisely so the bad reading loses the vocabulary it was hiding in: a consumer that + wants "where is this symbol" uses the span, and a consumer that wants the atom's content uses + the stored `text`. Rendering the span and calling it the atom renders, for the shell, the whole + file.""" + + path: str + blob_sha: str + layer: str + chunk_index: int + content_id: str + slot: str + slot_line_start: int + slot_line_end: int + current: bool = False + tombstoned: bool = False + + +@dataclass(frozen=True) +class CurrencyReport: + """What D2 step 4 actually changed. Both counts are rows whose flag MOVED, so a converged + re-land reports (0, 0) — the honest reading of "idempotent by convergence".""" + + made_current: int = 0 + superseded: int = 0 + + +@dataclass(frozen=True) +class PurgeReport: + """The recorded hole (D5). A purge that silently no-ops satisfies "never deletes" vacuously, + so both halves are counted and a caller can assert the hole EXISTS.""" + + vector_rows_deleted: int = 0 + memberships_tombstoned: int = 0 + + +_DDL = """ +CREATE TABLE IF NOT EXISTS memberships ( + path TEXT NOT NULL, -- the occupancy's file path (mutable: a rename is a NEW + -- fiber over the SAME atoms, never a re-hash — D0) + blob_sha TEXT NOT NULL, -- the git blob sha: the version's identity + layer TEXT NOT NULL, -- code_ast (L0a) / code_text (L0b) / codedoc (L1) + chunk_index INTEGER NOT NULL, -- position in the PURE derivation's chunk list, so two + -- identical windows stay two rows (the F5 multiset pin) + content_id TEXT NOT NULL, -- the atom: "{layer}:{content_hash}" (path-free, D1) + slot TEXT NOT NULL, -- qualname for L0a ONLY; '' for L0b/L1 (R4, no re-slot) + slot_line_start INTEGER NOT NULL, -- the SLOT's declared extent, NOT the atom's coverage (A2) + slot_line_end INTEGER NOT NULL, + current INTEGER NOT NULL DEFAULT 0, -- is this the path's HEAD projection? (D2 step 4) + tombstoned INTEGER NOT NULL DEFAULT 0, -- a purged atom's recorded hole (D5) + PRIMARY KEY (path, blob_sha, layer, chunk_index) +); +CREATE INDEX IF NOT EXISTS memberships_content ON memberships(content_id); +CREATE INDEX IF NOT EXISTS memberships_fiber ON memberships(path, blob_sha); +CREATE INDEX IF NOT EXISTS memberships_path ON memberships(path); + +-- The atom ledger: which embedder geometry each landed atom's vector came from. NOT a second +-- truth about occupancy (PD-3's fence is about lineage edges, and this stores none) — it records +-- the one fact the vector table structurally CANNOT: its Arrow schema is shared with the prose +-- lane and has no embedder column, and a stored vector's `len()` gives back `dim` but never +-- `model`. Without it the embedder pin above is unenforceable, and a model change would silently +-- reuse the old geometry forever. +CREATE TABLE IF NOT EXISTS atoms ( + content_id TEXT PRIMARY KEY, + layer TEXT NOT NULL, + embedder_model TEXT NOT NULL, + embedder_dim INTEGER NOT NULL, + landed_at TEXT NOT NULL +); +""" + + +@dataclass +class MembershipStore: + """The occupancy relation, in SQLite beside the vault catalog — outer ring, never imported by + the kernel. + + Reads are cheap point/range queries over the two indexes; the whole surface is deliberately + small, because everything history-shaped (runs, edges, forks, joins) is DERIVED from these rows + rather than stored (PD-3: one source of truth for lineage).""" + + path: Path + + def __post_init__(self) -> None: + if str(self.path) != ":memory:": + self.path.parent.mkdir(parents=True, exist_ok=True) + self._conn = sqlite3.connect(str(self.path)) + self._conn.row_factory = sqlite3.Row + self._conn.executescript(_DDL) + self._conn.commit() + + # ── the atom ledger: presence keyed to (layer, content_hash) AND the embedder ──────────── + + def known_atoms(self, content_ids: Iterable[str], embedder: EmbedderIdentity) -> set[str]: + """Which of `content_ids` are already in the plane UNDER THIS EMBEDDER (D2 step 2). + + The embedder half is the whole point: an atom landed by another model is present as a row + but not usable as a reuse, because reusing it would put two geometries in one ANN space. + A suite that only ever exercises one embedder cannot see the difference, which is why the + pin carries its own test case.""" + wanted = list(dict.fromkeys(content_ids)) + if not wanted: + return set() + out: set[str] = set() + for start in range(0, len(wanted), 400): + batch = wanted[start:start + 400] + marks = ", ".join("?" for _ in batch) + rows = self._conn.execute( + f"SELECT content_id FROM atoms WHERE content_id IN ({marks}) " + "AND embedder_model = ? AND embedder_dim = ?", + [*batch, embedder.model, embedder.dim]).fetchall() + out.update(str(r["content_id"]) for r in rows) + return out + + def record_atoms(self, atoms: Iterable[tuple[str, str]], + embedder: EmbedderIdentity) -> int: + """Record `(content_id, layer)` pairs as landed under `embedder`. Returns rows written. + + A re-land under a CHANGED embedder overwrites the identity (the atom's vector really was + re-embedded), so the ledger always describes the geometry currently in the plane.""" + rows = [(cid, layer, embedder.model, embedder.dim, _utcnow()) + for cid, layer in dict.fromkeys(atoms)] + if not rows: + return 0 + self._conn.executemany( + "INSERT INTO atoms (content_id, layer, embedder_model, embedder_dim, landed_at) " + "VALUES (?, ?, ?, ?, ?) ON CONFLICT(content_id) DO UPDATE SET " + "layer = excluded.layer, embedder_model = excluded.embedder_model, " + "embedder_dim = excluded.embedder_dim, landed_at = excluded.landed_at", rows) + self._conn.commit() + return len(rows) + + def ledger_atom_ids(self) -> set[str]: + return {str(r["content_id"]) + for r in self._conn.execute("SELECT content_id FROM atoms").fetchall()} + + def forget_atom(self, content_id: str) -> int: + """Drop one atom from the ledger — the purge's other half (D5). Never called by anything + else: nothing else removes geometry.""" + cur = self._conn.execute("DELETE FROM atoms WHERE content_id = ?", [content_id]) + self._conn.commit() + return int(cur.rowcount) + + # ── fibers: a version's complete projection ───────────────────────────────────────────── + + def write_fiber(self, rows: Sequence[Membership]) -> int: + """Write a version's fiber (D2 step 3). Returns rows actually inserted. + + `INSERT OR IGNORE`: an existing fiber's rows STAND, because derivation is pure so the + re-derived fiber is equal by construction. Returning 0 is therefore the correct, expected + answer on a re-land — and is precisely why it must NOT be read as "nothing to do": the + caller still owes step 4. One transaction, so a crash leaves the fiber whole or absent, + never half (D8: SQLite is the reference truth).""" + if not rows: + return 0 + before = self.count() + self._conn.executemany( + "INSERT OR IGNORE INTO memberships (path, blob_sha, layer, chunk_index, content_id, " + "slot, slot_line_start, slot_line_end, current, tombstoned) " + "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)", + [(m.path, m.blob_sha, m.layer, m.chunk_index, m.content_id, m.slot, + m.slot_line_start, m.slot_line_end, int(m.current), int(m.tombstoned)) + for m in rows]) + self._conn.commit() + return self.count() - before + + def fiber(self, path: str, blob_sha: str) -> list[Membership]: + """One version's complete projection, in `chunk_index` order — i.e. in the order the pure + derivation emitted the chunks, so a fiber read back IS the version's chunk list.""" + return [_row(r) for r in self._conn.execute( + "SELECT * FROM memberships WHERE path = ? AND blob_sha = ? ORDER BY chunk_index", + [path, blob_sha]).fetchall()] + + def blobs_of(self, path: str) -> list[str]: + """Every blob this path has a fiber for, oldest-recorded first (by insertion rowid).""" + return [str(r["blob_sha"]) for r in self._conn.execute( + "SELECT blob_sha FROM memberships WHERE path = ? GROUP BY blob_sha ORDER BY MIN(rowid)", + [path]).fetchall()] + + def fibers(self) -> list[tuple[str, str]]: + return [(str(r["path"]), str(r["blob_sha"])) for r in self._conn.execute( + "SELECT path, blob_sha FROM memberships GROUP BY path, blob_sha " + "ORDER BY path, MIN(rowid)").fetchall()] + + def fiber_sizes(self) -> dict[tuple[str, str], int]: + return {(str(r["path"]), str(r["blob_sha"])): int(r["n"]) for r in self._conn.execute( + "SELECT path, blob_sha, count(*) AS n FROM memberships GROUP BY path, blob_sha" + ).fetchall()} + + def count(self) -> int: + """|M| — the total number of occupancies.""" + row = self._conn.execute("SELECT count(*) FROM memberships").fetchone() + return int(row[0]) if row else 0 + + # ── currency reconciliation: D2 step 4, the C1 case ───────────────────────────────────── + + def reconcile_currency(self, path: str, head_blob_sha: str) -> CurrencyReport: + """Set `current=1` on exactly the fiber whose blob is the path's HEAD, and `current=0` on + the path's every other fiber. **NEVER skipped, even when the fiber write was a no-op.** + + This is the whole reason a re-land is idempotent. The tempting shortcut — "the fiber + already exists, so return" — is the C1 bug in its exact original form: land A, land B, land + A again; A's fiber exists with `current=false`, the short-circuit skips this call, and the + store is left claiming **B** is HEAD. Nothing raises, nothing logs, and every default + (current-view) read is now wrong. Convergence, not short-circuiting. + + Idempotent by construction: the counts are of rows whose flag actually MOVED, so a second + call reports (0, 0) and writes nothing.""" + cur = self._conn.execute( + "SELECT count(*) FROM memberships WHERE path = ? AND blob_sha = ? AND current = 0", + [path, head_blob_sha]).fetchone() + stale = self._conn.execute( + "SELECT count(*) FROM memberships WHERE path = ? AND blob_sha != ? AND current = 1", + [path, head_blob_sha]).fetchone() + made = int(cur[0]) if cur else 0 + dropped = int(stale[0]) if stale else 0 + if made: + self._conn.execute( + "UPDATE memberships SET current = 1 " + "WHERE path = ? AND blob_sha = ? AND current = 0", [path, head_blob_sha]) + if dropped: + self._conn.execute( + "UPDATE memberships SET current = 0 WHERE path = ? AND blob_sha != ? " + "AND current = 1", [path, head_blob_sha]) + if made or dropped: + self._conn.commit() + return CurrencyReport(made_current=made, superseded=dropped) + + # ── the frequency plane (D6): two counts, never conflated ──────────────────────────────── + + def n_doc(self, content_id: str, *, current_only: bool = True) -> int: + """Distinct PATHS holding this atom — the document-frequency reading, immune to + within-file repetition. `current_any(v) ⇔ n_doc(v, t) > 0` is the carried invariant (R3).""" + clause = "AND current = 1" if current_only else "" + row = self._conn.execute( + f"SELECT count(DISTINCT path) FROM memberships " + f"WHERE content_id = ? AND tombstoned = 0 {clause}", [content_id]).fetchone() + return int(row[0]) if row else 0 + + def n_occ(self, content_id: str, *, current_only: bool = True) -> int: + """Membership ROWS holding this atom — the multiset reading, where L0b's repeated windows + count. Never conflated with `n_doc` (the F5 pin: they are different questions).""" + clause = "AND current = 1" if current_only else "" + row = self._conn.execute( + f"SELECT count(*) FROM memberships " + f"WHERE content_id = ? AND tombstoned = 0 {clause}", [content_id]).fetchone() + return int(row[0]) if row else 0 + + def atom_ids_of_path(self, path: str) -> set[str]: + return {str(r["content_id"]) for r in self._conn.execute( + "SELECT DISTINCT content_id FROM memberships WHERE path = ?", [path]).fetchall()} + + def currently_held(self, content_ids: Iterable[str]) -> set[str]: + """Which of `content_ids` have at least one current, un-tombstoned occupancy — i.e. the + set whose `current_any` should be true. One query, so D2 step 5's crossing arithmetic is + cheap even when a path's history is long.""" + wanted = list(dict.fromkeys(content_ids)) + if not wanted: + return set() + out: set[str] = set() + for start in range(0, len(wanted), 400): + batch = wanted[start:start + 400] + marks = ", ".join("?" for _ in batch) + rows = self._conn.execute( + f"SELECT DISTINCT content_id FROM memberships WHERE content_id IN ({marks}) " + "AND current = 1 AND tombstoned = 0", batch).fetchall() + out.update(str(r["content_id"]) for r in rows) + return out + + # ── the read path (D3): one atom resolves to all its occupancies ───────────────────────── + + def occupancies(self, content_id: str, *, + include_superseded: bool = False) -> list[Membership]: + """Where this atom lives. One atom may resolve to SEVERAL occupancies — that is the + feature, not a defect: a hit natively answers "this idea lives in versions v3–v7 of X and + also in Y". Default consumers see current occupancies only (D3).""" + clause = "" if include_superseded else "AND current = 1" + return [_row(r) for r in self._conn.execute( + f"SELECT * FROM memberships WHERE content_id = ? {clause} " + "ORDER BY path, blob_sha, layer, chunk_index", [content_id]).fetchall()] + + # ── derived lineage (PD-3: never a second stored truth) ───────────────────────────────── + + def slot_runs(self, path: str, chain: Sequence[str]) -> dict[str, list[str]]: + """Per slotted `(path, slot)`, the occupancy chain collapsed into RUNS of equal occupants + along `chain` (the path's FIRST-PARENT blob sequence, oldest first). + + **ADJACENT collapse, never distinct-collapse** — this is the C1 formulation and the whole + reason §4 was re-keyed on runs: A → B → A gives runs `[A, B, A]`, three runs and two edges, + so a revert stays visible. Collapsing distinct occupants would report one edge and quietly + erase the revert, which is the same failure `core/stores/versions.py` documents at note + grain. `ops/code_lineage.py:151-156` already collapses only adjacent repeats at file grain, + so the two grains agree. + + `chain` is handed in rather than read from this store because chain MEMBERSHIP is not a + membership fact: chains are first-parent while the ledger walks all commits, so a + side-branch fiber is a real member of M that sits on NO chain (D4/F3). Quantifying over + every fiber instead of over chain members is the error this signature makes hard. + + **Slotted means L0a — the LAYER, not a non-empty name** (D1/R4). L0b and L1 carry + `qualname=''` as built and are membership-only, honestly chainless; the L0a MODULE SHELL + also carries `''`, but that empty string is a real slot (the shell is a symbol-shaped + region of the file) and it chains like any other. Reading slottedness off `slot != ''` + would silently drop the shell's lineage — the one slot every file has.""" + runs: dict[str, list[str]] = {} + for blob in chain: + for m in self.fiber(path, blob): + if m.layer != LAYER_CODE_AST: # L0b/L1 are membership-only, chainless (R4) + continue + seq = runs.setdefault(m.slot, []) + if not seq or seq[-1] != m.content_id: + seq.append(m.content_id) + return runs + + def slot_edges(self, path: str, chain: Sequence[str]) -> dict[str, list[tuple[str, str]]]: + """The supersession edges: consecutive-run pairs per slot, so per slot + `|edges| = |runs| - 1` by construction. Edge identity is + `(path, slot, old_hash -> new_hash, at blob transition)`; the endpoints differ by + construction because runs collapse equal adjacent occupants.""" + return {slot: list(zip(seq, seq[1:], strict=False)) + for slot, seq in self.slot_runs(path, chain).items()} + + # ── D5: purge — the ONE removal, and it leaves a recorded hole ─────────────────────────── + + def tombstone_atom(self, content_id: str) -> int: + """Mark every occupancy of a purged atom as a hole. Returns rows tombstoned.""" + cur = self._conn.execute( + "UPDATE memberships SET tombstoned = 1, current = 0 " + "WHERE content_id = ? AND tombstoned = 0", [content_id]) + self._conn.commit() + return int(cur.rowcount) + + # ── D8: the repair pass — what makes a crash a non-event ──────────────────────────────── + + def orphan_atom_ids(self) -> set[str]: + """Atoms in the plane with NO occupancy at all — dormant geometry from a land that crashed + between the vector insert and the fiber write (D8's named window). + + The state is deliberately OBSERVABLE rather than merely harmless: "nothing dangles by + reference" is true of a store that never wrote anything, so a crash test that cannot see + the orphan proves nothing about repair. Nothing here deletes them — the idempotent re-land + adopts them at zero embed cost, which is exactly why the write order puts vectors first.""" + return {str(r["content_id"]) for r in self._conn.execute( + "SELECT content_id FROM atoms WHERE content_id NOT IN " + "(SELECT DISTINCT content_id FROM memberships)").fetchall()} + + def close(self) -> None: + self._conn.close() + + +def _row(r: sqlite3.Row) -> Membership: + return Membership( + path=str(r["path"]), blob_sha=str(r["blob_sha"]), layer=str(r["layer"]), + chunk_index=int(r["chunk_index"]), content_id=str(r["content_id"]), slot=str(r["slot"]), + slot_line_start=int(r["slot_line_start"]), slot_line_end=int(r["slot_line_end"]), + current=bool(r["current"]), tombstoned=bool(r["tombstoned"]), + ) + + +# ── cross-store operations: the read join, the repair pass, and the purge ──────────────────── + +@dataclass(frozen=True) +class ResolvedHit: + """One ANN hit, resolved (D3). `row` is the atom — geometry, `text`, `_distance` if the hit came + from a search — and `occupancies` are the places it lives. + + ⚑ `occupancies` carry the SLOT's extent, not the atom's text coverage (A2). A consumer + rendering `slot_line_start..slot_line_end` as "this atom" is wrong, and maximally wrong for the + module shell, whose extent is the whole file: the atom's content is `row["text"]`.""" + + row: dict[str, object] + occupancies: tuple[Membership, ...] + + +def resolve_occupancies(memberships: MembershipStore, hits: Sequence[dict[str, object]], *, + include_superseded: bool = False) -> list[ResolvedHit]: + """The D3 read-path join: top-k atoms -> their occupancies `(path, blob, slot, lines, ...)`. + + Rows that are not shed atom rows (note rows, and any pre-D1 per-version code row) resolve to no + occupancy and keep their own coordinates — this join adds a lane, it does not take one away.""" + out: list[ResolvedHit] = [] + for h in hits: + cid = str(h.get("id", "")) + occ = (tuple(memberships.occupancies(cid, include_superseded=include_superseded)) + if is_code_atom_row(dict(h)) else ()) + out.append(ResolvedHit(row=h, occupancies=occ)) + return out + + +def repair_current_any(vectors: VectorStore, memberships: MembershipStore) -> tuple[int, int]: + """Rebuild the `current_any` cache from membership truth (D8/R3). Returns (raised, lowered). + + `current_any` is a cache with a rebuild path, and this is the rebuild path: the flag is + recomputed as `n_doc(v, t) > 0` over every atom in the plane, and only the DRIFTED rows are + written. Called after a crash, or on cadence as the R3 ratchet.""" + rows = vectors.atom_rows() + ids = [str(r["id"]) for r in rows] + held = memberships.currently_held(ids) + to_true = [str(r["id"]) for r in rows if not bool(r.get("current")) and str(r["id"]) in held] + to_false = [str(r["id"]) for r in rows if bool(r.get("current")) and str(r["id"]) not in held] + return (vectors.set_current_any(to_true, True), + vectors.set_current_any(to_false, False)) + + +def current_any_drift(vectors: VectorStore, memberships: MembershipStore) -> list[str]: + """Atom ids where the lance flag disagrees with membership truth — the R3 ratchet's reading, + and the `current_any(v) ⇔ n_doc(v, t) > 0` invariant as a checkable list (empty ⇒ holds).""" + rows = vectors.atom_rows() + held = memberships.currently_held([str(r["id"]) for r in rows]) + return sorted(str(r["id"]) for r in rows + if bool(r.get("current")) != (str(r["id"]) in held)) + + +def purge_atom(vectors: VectorStore, memberships: MembershipStore, + content_id: str) -> PurgeReport: + """The ONE removal (D5, finding-0164 — owner-gated; privacy outranks lineage). + + Delete the vector row, tombstone its memberships, forget its ledger entry: a RECORDED HOLE, + never a silent one. Both counts are returned because "no code path deletes a vector" is + satisfied vacuously by a purge that does nothing — the criterion is that the hole EXISTS.""" + deleted = vectors.delete_atom(content_id) + tombstoned = memberships.tombstone_atom(content_id) + memberships.forget_atom(content_id) + return PurgeReport(vector_rows_deleted=deleted, memberships_tombstoned=tombstoned) + + +def open_membership_store(config: Config | None = None) -> MembershipStore: + """The configured membership store — SQLite beside the vault catalog, exactly as + `open_version_store` sites the version history.""" + from core.kernel.config import get_config + + cfg = config or get_config() + return MembershipStore(cfg.paths.derived_store.parent / "memberships.sqlite") diff --git a/core/stores/vectorstore.py b/core/stores/vectorstore.py index 9623482a..6047d84f 100644 --- a/core/stores/vectorstore.py +++ b/core/stores/vectorstore.py @@ -23,6 +23,10 @@ TABLE = "chunks" +# How many ids one `id IN (...)` predicate may name before `set_current_any` starts a new batch. +# Bounds the SQL string, never the semantics: the batches are independent in-place updates. +_ID_PREDICATE_CHUNK = 400 + # The layer coordinate (dn-code-ingest-pipeline §2.2): note rows default 'prose'; the code embed # lane discriminates its three projections. One shared table, one additive schema — the fiber @@ -50,12 +54,18 @@ def _schema(dim: int) -> pa.Schema: ("qualname", pa.string()), # code fiber: the symbol ('' for note/L0b/module-shell rows) ("line_start", pa.int32()), # code fiber: first source line (0 for note rows) ("line_end", pa.int32()), # code fiber: last source line (0 for note rows) - # The temporal-corpus flag (dn-temporal-code-corpus D2/D3, bp-099): is this row part of the - # source path's CURRENT (HEAD) projection? A superseded code version is RETAINED with - # current=false (keep-and-link, never deleted); note rows carry the vacuous current=true. - # Additive: a pre-bp-099 store (no `current`) is migrated in place by - # `_migrate_current_if_needed` (every row stamped current=true — correct while the store - # holds only HEAD code + note rows, which is why the migration lands BEFORE any backfill). + # [banner: correction] The temporal-corpus flag (dn-temporal-code-corpus D2/D3, bp-099) + # WAS one reading: is this row part of the source path's CURRENT (HEAD) projection? Under + # dn-vector-membership-store D1 (bp-152) this ONE column now carries TWO readings, because + # the two lanes store different objects under one schema: + # * NOTE rows (and any pre-D1 per-version code row) keep the old reading — the vacuous + # current=true for prose, keep-and-link current=false for a superseded code version. + # * CODE-ATOM rows read it as **`current_any`**: does ANY current membership contain this + # atom? An atom is not a version, so "is this the HEAD projection" is not a question it + # can answer; `current_any` is the cheap ANN prefilter over membership truth (D3), and + # the membership SQLite store is the reference truth it caches (D8, R3). + # Both readings are served by the ONE `current = true` prefilter clause in `search` — which + # is exactly why D1 re-reads the column instead of adding a second one. ("current", pa.bool_()), ("vector", pa.list_(pa.float32(), dim)), ]) @@ -67,6 +77,30 @@ def _schema(dim: int) -> pa.Schema: "layer": LAYER_PROSE, "qualname": "", "line_start": 0, "line_end": 0, "current": True, } +# The occupancy columns a CODE-ATOM row sheds (dn-vector-membership-store D1, bp-152). They are +# shed from the ROW, never from the schema — note rows still carry them, so every prose-lane +# consumer is untouched. The empty values are load-bearing, not cosmetic: `delete_source` / +# `index_amendment` are `source_path`-scoped and can never match `''` (the note's D1 shed-safety +# claim, verified at `core/ingest/index.py:87`), and `is_code_atom_row` reads them as the shed +# marker. `provenance` is NOT here — it STAYS, because the mirror firewall is a row PREFILTER +# (`search`, below) and shedding the column would not weaken the firewall, it would REMOVE it. +ATOM_ROW_SHED: dict[str, Any] = { + "digest": "", "source_path": "", "chunk_index": 0, "qualname": "", + "line_start": 0, "line_end": 0, +} + + +def is_code_atom_row(row: dict[str, Any]) -> bool: + """Is this row a shed CODE **atom** row (D1) rather than a source-object chunk row? + + An atom row is geometry with no occupancy: its occupancy lives in the membership relation + (`core/stores/memberships.py`), so it carries no `source_path` and no `digest`. This predicate + is the ONE place that reading is spelled out, so `all_rows`' structural guard and the + membership store's repair pass cannot drift apart.""" + return (row.get("provenance") == Provenance.CODE.value + and not str(row.get("source_path") or "") + and not str(row.get("digest") or "")) + def _sql_str(value: str) -> str: """One arbitrary string as a SQL literal for a LanceDB predicate — single quotes doubled. @@ -287,19 +321,92 @@ def relabel_provenance(self, old: str, new: str) -> int: return len(rows) def all_rows(self, *, - provenances: Iterable[Provenance] | None = None) -> list[dict[str, Any]]: + provenances: Iterable[Provenance] | None = None, + include_atom_rows: bool = False) -> list[dict[str, Any]]: """Full scan, optionally restricted to provenance classes — the read the dreaming agent clusters over. The clustering itself is deterministic and model-free (§9), so the mirror passes `provenances={AUTHORED}` and observed exhaust never seeds a dream. - Single-user corpus scale; filtered in Python after the Arrow scan for portability.""" + Single-user corpus scale; filtered in Python after the Arrow scan for portability. + + [banner: correction] **The shed-atom guard (bp-152 Item 3, the note's §3 Q5 gap).** An + UNSCOPED read (`provenances=None`) now excludes shed CODE-atom rows unless the caller asks + for them by name. The reason is a silent corruption, not tidiness: every structural + group-by-`digest` consumer keys on a column an atom row does not have. + `core/kernel/stores/sourceset.py`'s `source_sets(store)` defaults to ALL strata by design + ("a structural grouping utility, not a mirror read"), and `group_sources` keys on + `r["digest"]` — so shed rows, all carrying `digest=''`, would collapse into ONE bogus + SourceSet keyed `''`. `MixedProvenanceError` cannot catch it (it needs a digest spanning + *several* provenances; these rows are uniformly CODE), so it fails with no visible error at + all. `core/curator/curator.py`'s `prune_candidates` has the identical shape and would + report the whole atom plane as one orphaned digest. + + The guard is here rather than in `sourceset` deliberately: the kernel module may never + learn about the outer-ring membership store (the C5/D3 ring pin — an import would demote + `sourceset` from the inner-ring fixed point), so the shed side owns the consequence of its + own shed. Excluding is the fail-closed half of the note's "excludes them or raises": a + structural grouping over geometry-without-occupancy is not a smaller answer, it is a wrong + one, and a code consumer reaches source objects through membership fibers (D3) — never + through group-by-digest. + + A caller that genuinely wants the whole physical table passes `include_atom_rows=True`; + a caller that wants the code lane passes `provenances={CODE}`, which is how + `CodeCorpusSync` and the eval battery already read it, and is unaffected.""" if TABLE not in self._db.list_tables().tables: return [] rows = self._table().to_arrow().to_pylist() if provenances is None: - return rows + if include_atom_rows: + return rows + return [r for r in rows if not is_code_atom_row(r)] allowed = {Provenance(p).value for p in provenances} return [r for r in rows if r.get("provenance") in allowed] + def atom_rows(self) -> list[dict[str, Any]]: + """Every shed CODE-atom row (D1) — the vector plane V, as rows. `len(atom_rows())` is |V| + for the append-only invariant (§4: no test may ever observe |V| decrease except across a + logged purge).""" + return [r for r in self.all_rows(provenances={Provenance.CODE}) if is_code_atom_row(r)] + + def set_current_any(self, ids: Iterable[str], value: bool) -> int: + """Set `current_any` on exactly the named atom rows (D2 step 5). Returns rows written. + + `current` is a lance column, so a flip rewrites fragments — which is why D2 step 5 and the + note's §3 pin say *only the atoms whose current-membership count crossed 0↔1*. This method + takes the crossing set and nothing else; computing it is the lander's job (it is pure + membership arithmetic and belongs in SQLite, not here). An empty set writes nothing. + + Predicates are chunked so a large crossing set cannot build an unbounded SQL string.""" + wanted = list(dict.fromkeys(ids)) + if not wanted or TABLE not in self._db.list_tables().tables: + return 0 + table = self._table() + written = 0 + for start in range(0, len(wanted), _ID_PREDICATE_CHUNK): + batch = wanted[start:start + _ID_PREDICATE_CHUNK] + where = f"id IN ({', '.join(_sql_str(i) for i in batch)})" + # portable count (do NOT rely on UpdateResult — bp-103 §11), then one in-place update + n = table.count_rows(where) + if n: + table.update(where, {"current": value}) + written += n + return written + + def delete_atom(self, content_id: str) -> int: + """Delete one atom row by its `(layer, content_hash)` id — the D5 PURGE path, and the ONLY + machinery in this module that removes a vector row (finding-0164; owner-gated, privacy + outranks lineage). Returns rows deleted, so a purge that silently no-ops is observable: + "never deletes" is satisfied vacuously by a purge that does nothing, so the count is the + thing a test asserts on. The membership half of the hole (tombstoning) is the membership + store's — see `core.stores.memberships.purge_atom`, which owns both halves.""" + if TABLE not in self._db.list_tables().tables: + return 0 + table = self._table() + where = f"id = {_sql_str(content_id)}" + n = table.count_rows(where) + if n: + table.delete(where) + return n + def search(self, vector: list[float], *, k: int = 5, provenances: Iterable[Provenance] | None = None, include_superseded: bool = False) -> list[dict[str, Any]]: @@ -309,7 +416,14 @@ def search(self, vector: list[float], *, k: int = 5, Default retrieval is CURRENT-VIEW (dn-temporal-code-corpus D3, bp-099): a superseded code version (`current=false`, kept by keep-and-link) never surfaces unasked, so every existing consumer is unchanged (note rows carry the vacuous current=true). A temporal consumer opts - into history with `include_superseded=True`.""" + into history with `include_superseded=True`. + + Under D1 (bp-152) the same clause serves the atom plane with the `current_any` reading — + "does ANY current membership contain this atom" — so an atom whose every occupancy is + superseded drops out of the default search exactly as a superseded version used to. This + search returns ATOMS; resolving each hit to its occupancies `(path, blob, slot, lines)` is + the membership join (D3, `core.stores.memberships.resolve_occupancies`), and the join is + where a code consumer learns *where* a hit lives.""" if TABLE not in self._db.list_tables().tables: return [] q = self._table().search(vector).metric("cosine") diff --git a/docs/build-plans/bp-152/journal.md b/docs/build-plans/bp-152/journal.md index 0393363d..db27a8e5 100644 --- a/docs/build-plans/bp-152/journal.md +++ b/docs/build-plans/bp-152/journal.md @@ -88,3 +88,144 @@ - **Depends on bp-151.** The dedup claims are only true once identity survives a rename. Confirm bp-151 has merged before starting. + +--- + +## Session 1 — 2026-08-08 · builder, worktree `build/bp-152-membership-store` (base 68d8d39) + +**Status: all eight items built; §8(a)–(f) green and non-vacuous. One issue filed for a +consequence the plan did not enumerate (the eval harness reach, below).** + +### What was built, item by item + +**Item 1 — the atom row.** `code_rows(chunks, vectors, *, current=False)`. The signature lost +`path` and `blob_sha` because nothing on an atom row uses them any more: the id is +`atom_id(chunk) = f"{layer}:{content_hash}"` (path-free, corpus-wide) and every occupancy column +is shed to its empty value through the new `ATOM_ROW_SHED` dict in `vectorstore.py`. +`provenance` stays, hardcoded, and `test_every_atom_row_keeps_its_provenance_and_sheds_its_occupancy` +asserts *presence and equality*, not merely "no code leaked" — the firewall is a prefilter, so an +empty column removes it rather than weakening it. + +**One micro-gap in D1, decided and recorded rather than inferred silently.** The note's shed list +enumerates `source_path, digest, qualname, line_*, chunk_index`; the keep list is +`id, layer, text, vector, provenance, current`. **`title` is in neither.** On a code row `title` +was set to `path` — it *is* `source_path` under another name — so it is shed with the occupancy +columns. Keeping it would stamp every shared atom with its first-landed path, which is precisely +the coordinate D0's consequence note says must never be relied on. Flagged here and in the PR body +for the reviewer; it is a one-line reversal if the owner reads D1's enumeration as exhaustive. + +**Item 2 — the store.** `core/stores/memberships.py`, outer ring, sqlite. Key +`(path, blob_sha, layer, chunk_index)` exactly as pinned. `code_memberships()` in +`code_corpus.py` is the A2 translation point and builds **its own list** — one row per chunk, +never per distinct atom — so the atom-side `by_id.setdefault` collapse cannot be inherited. + +The A2 rename landed on both halves in one change: `CodeChunk.slot_line_start`/`slot_line_end` +and `memberships.slot_line_start`/`slot_line_end`, with the vector-row Arrow keys untouched +(A2.3 — shared with the prose lane). + +**A second table lives in the same SQLite file: `atoms`.** It records `(content_id, layer, +embedder_model, embedder_dim, landed_at)`. This is not a second truth about occupancy — it stores +the one fact the vector table structurally *cannot*: its Arrow schema is shared with the prose +lane and has no embedder column, and a stored vector's `len()` recovers `dim` but never `model`. +Without it the owner's embedder pin is unenforceable. It is also what makes an orphan observable +(§8(e)): an atom in the ledger with no membership row. + +**Item 3 — the guard.** It lives in `VectorStore.all_rows`: an **unscoped** read +(`provenances=None`) now excludes shed atom rows unless the caller passes +`include_atom_rows=True`. No kernel edit; `sourceset.py` untouched. The test reproduces the +falsifier first — `group_sources(vs.all_rows(include_atom_rows=True))` really does return the +bogus set keyed `''`, containing every atom — and only then asserts the guard. A second test pins +that the guard keys on the **shed** (`is_code_atom_row`), never on the provenance, so the live +store's existing pre-D1 code rows keep grouping until bp-153 rebuilds them. + +⚠ **One consequence worth the reviewer's eye:** `core/ingest/index.py`'s `rekey_store` reads +`all_rows()` unscoped, then `reset()`s and re-adds. With the guard it would *drop* shed atom rows; +without the guard it would *corrupt* them (it recomputes ids from `source_path` + `text`, and an +atom row's `source_path` is `''`). It was already broken by the shed either way, it is a one-shot +historical migration behind `scripts/migrate_chunk_keys.py`, and `core/ingest/index.py` is a +note-lane path deliberately out of this plan's scope (PD-2/R5). Filed, not fixed here. + +**Item 4 — `land()`.** `CodeLander.land(path, blob_sha, chunks, *, head_blob_sha=None)` in +`code_corpus.py`, all five steps in the D8 order. Step 4 runs unconditionally. + +Step 5's crossing set is computed as pure membership arithmetic, with **no read of the lance +column**: `candidates = set(atom ids) | atom_ids_of_path(path)`, `before = currently_held(candidates)` +taken *before* the fiber write, `after` taken after reconciliation, and only +`after - before` / `before - after` are flipped. Two consequences worth writing down: (i) a fresh +atom is inserted with `current=False` (at insert time, under D8's write order, it genuinely has no +occupancy), so its 0→1 crossing is picked up in step 5; (ii) an atom already current in another +path is inside `candidates` via the id list, so it is not redundantly rewritten. + +`reconcile(path, head)` and `supersede_path(path)` are steps 4–5 alone. **`sync()` calls +`reconcile` for every unchanged HEAD path** — skipping it is the C1 short-circuit one level up: +after A→B→A the HEAD fiber exists, the file reads as unchanged, and B stays current forever. Two +counting queries per path is not a price worth trading for that. + +`CodeCorpusSync` gained `memberships` and `embedder_identity` as **required** fields. Optional +would have been a footgun of exactly the kind the "wiring is part of finishing" rule names: a +lander with no occupancy record is not a lander, and a reuse decision with no embedder identity is +the geometry-mixing bug waiting to happen. Its D-fiber state re-homed from +`{(source_path, digest)}` to `memberships.fibers()` — the note's §6 re-home (1), applied at this +call site only. **The daemon's incompleteness probe (`ops/lifecycle/launcher.py:387`) still reads +the old shape and is bp-153's, exactly as the plan says; it will false-positive against a +rebuilt store, which is a bp-153 precondition, not a regression introduced here.** + +**Items 5/6/7** — read-path join (`resolve_occupancies`, `ResolvedHit`), purge +(`purge_atom` → deletes the row, tombstones the memberships, forgets the ledger entry), the D8 +repair pass (`orphan_atom_ids`, `repair_current_any`, `current_any_drift`), and derived lineage +(`slot_runs`/`slot_edges`, **adjacent** collapse, quantified over a chain handed in). + +One correction to the plan's phrasing, found while building: "slotted" is the **layer**, not a +non-empty name. The L0a **module shell** carries `qualname=''` just like L0b/L1, so reading +slottedness off `slot != ''` would silently drop the lineage of the one slot every file has. +`slot_runs` filters on `layer == LAYER_CODE_AST`. + +**Item 8 — surface repair.** Repaired: `test_code_corpus`, `test_code_retrieval`, +`test_code_mirror`, `test_code_vector_isolation`, `test_sourceset`, and — not in the plan's §5 +list — `test_code_lineage` (it constructs `CodeCorpusSync` and asserts stored +`(source_path, digest)`, so the shed moved it). No firewall assertion was relaxed; the +`test_code_mirror` / `test_code_vector_isolation` claims are unchanged in kind and now read +through the join. + +### The one reach beyond write_scope, and why + +`eval/harness/code_retrieval.py`'s `ranked_paths` reads `h["source_path"]` off a hit. On atom rows +that is `''`, so the M-C3/M-C5 battery silently ranked nothing — the plan's §3 investigation did +not surface it (it is the D3 read path, one file outside the enumerated scope). `ranked_paths` and +`run_mc3` gained an **optional** `memberships` parameter: pass it and each hit contributes every +path it currently occupies, which is the honest reading of a shared atom; omit it and the pre-split +behavior is byte-identical. Additive, ~20 lines, and it keeps the instrument working rather than +leaving a known-broken one behind. Called out in the PR body for the merge audit. + +### How each criterion was made non-vacuous + +| criterion | the precondition that makes it bite | +|---|---| +| §8(a) revert | A and B are asserted DISTINCT (`\|V\|` grew between them) before the third land; the currency assertions carry the claim, and a **mutation test** exhibits the short-circuiting lander passing every count and leaving B current | +| §8(b) fork | `\|V\| == 3 < Σ chunks == 4` and `n_doc(shared) == 2` asserted first; the edit is hand-built so "exactly 1 new atom" is exact arithmetic (a real file's single L0b window recuts on any edit) | +| §8(c) purge | the note-lane removal paths are shown to ACT (the note row really is deleted) while `\|V\|` is unmoved; then the purge's report is asserted non-zero and the tombstoned rows are asserted to still EXIST | +| §8(d) retrieval | a superseded occupancy is asserted to exist AND an atom whose every home is superseded (`n_doc == 0`) — so the current-filter is not passing on an all-current store | +| §8(e) crash | the injection point is asserted: `\|V\|` grew, `\|M\|` did not, the fiber is empty, and `orphan_atom_ids()` is non-empty and equals the landed set | +| §8(f) invariants | one test asserts the fixture carries **all four shapes** before any invariant is read; the runs invariant asserts A→B→A gives 3 runs / 2 edges and that distinct-collapse would give 2; the fiber-sum invariant is broken by a subclass whose `fiber()` drops superseded rows (the fixture is asserted to hold one); the drift invariant is broken by a hand-flipped flag | +| Item 2 / A2 | the fixture is asserted to contain a nested symbol (`Foo.bar` inside `Foo`) and a non-empty module shell; the leaf symbol `top` is asserted to have span == coverage **exactly**, as the control that makes the strict-subset assertions mean something | +| embedder pin | its own test, with the same-embedder control asserted first so "re-embedded" means "the embedder changed", not "reuse never worked" | + +### Gate + +Recorded in the PR body. Baseline on the clean base (68d8d39) before any edit: +**5 failed, 2427 passed, 15 skipped** — exactly the three known-red classes (finding-0103 +self-containment ratchet, `test_dream_v2_live`, `test_worktree_enforcement` ×3, issue #13 / +finding-0280). + +### For bp-153 + +- The incompleteness probe (`ops/lifecycle/launcher.py:_code_backfill_incomplete`) must re-home to + `memberships.fibers()` **before** a rebuilt store exists, or it enqueues forever (finding-0166's + named falsifier). +- `repair_current_any` / `current_any_drift` are built and tested — they are the R3 ratchet's + instrument, ready to register as a gauge. +- `n_doc` / `n_occ` are built and separated (the F5 pin). `|M|` is `MembershipStore.count()`, `|V|` + is `len(VectorStore.atom_rows())` — the `|M|/|V|` gauge needs no new machinery. +- The carry-forward seed re-enters through `record_atoms(...)` under the live `EmbedderIdentity`; + an entry recorded under a different embedder is not a reuse hit, which is what makes the + 13,311-atom bulk reuse safe rather than a silent geometry mix. diff --git a/eval/harness/code_retrieval.py b/eval/harness/code_retrieval.py index 39a831fe..bc5bbd38 100644 --- a/eval/harness/code_retrieval.py +++ b/eval/harness/code_retrieval.py @@ -40,6 +40,7 @@ from core.ingest.embed import Embedder from core.ingest.index import semantic_search from core.kernel.provenance import MIRROR_READABLE, Provenance +from core.stores.memberships import MembershipStore from core.stores.vectorstore import ( LAYER_CODE_AST, LAYER_CODE_TEXT, @@ -78,14 +79,22 @@ def cosine(a: Sequence[float], b: Sequence[float]) -> float: # ── M-C3: retrieval quality — code lane vs docstring-only baseline ───────────────────────── def ranked_paths(query: str, embedder: _QueryEmbedder, store: VectorStore, *, - layers: Iterable[str], pool: int = _DEFAULT_POOL) -> list[tuple[str, float]]: + layers: Iterable[str], pool: int = _DEFAULT_POOL, + memberships: MembershipStore | None = None) -> list[tuple[str, float]]: """The ranked, path-deduplicated retrieval for one query restricted to `layers`. Pulls `pool` flat code chunks via an EXPLICIT `provenances={CODE}` search (never the mirror default — §7), filters to the requested layers, and collapses to one entry per source path at its best (lowest-distance) hit, preserving rank. Returns `[(path, distance)]` best-first. The Python-side layer filter + path dedup mirrors the store's single-user-scale posture - (`all_rows`/`rows_for_source`): pull a generous pool, refine in Python.""" + (`all_rows`/`rows_for_source`): pull a generous pool, refine in Python. + + [cross-ref: extension] Under the atom+membership split (dn-vector-membership-store D1, bp-152) + a code hit is an ATOM and carries no `source_path` — occupancy lives in the membership store, + and the D3 read path resolves it by JOIN. Pass `memberships` and each hit contributes every + path it currently occupies, which is the honest reading of a shared atom: one idea genuinely + living in two files ranks both, at the same distance. Without it the pre-split reading is + unchanged (rows that still carry `source_path` resolve to themselves), so this is additive.""" allowed = set(layers) hits = semantic_search(query, cast(Embedder, embedder), store, k=pool, provenances={Provenance.CODE}) @@ -94,16 +103,29 @@ def ranked_paths(query: str, embedder: _QueryEmbedder, store: VectorStore, *, for h in hits: if h.get("layer") not in allowed: continue - path = str(h["source_path"]) dist = float(h.get("_distance", 0.0)) - if path not in best: - best[path] = dist - order.append(path) - elif dist < best[path]: - best[path] = dist + for path in _hit_paths(h, memberships): + if path not in best: + best[path] = dist + order.append(path) + elif dist < best[path]: + best[path] = dist return [(p, best[p]) for p in order] +def _hit_paths(hit: dict[str, Any], memberships: MembershipStore | None) -> list[str]: + """Where one hit lives: its own `source_path` for a pre-split row, or its CURRENT occupancies + resolved through the membership join for a shed atom row (D3). An atom with no membership + resolves nowhere and contributes no path — correct, not silent: dormant geometry from an + interrupted land is not in any file yet.""" + own = str(hit.get("source_path") or "") + if own: + return [own] + if memberships is None: + return [] + return list(dict.fromkeys(m.path for m in memberships.occupancies(str(hit.get("id", ""))))) + + def _rank_of(paths: Sequence[tuple[str, float]], answers: frozenset[str]) -> int | None: """1-based rank of the first answer path in the ranking, or None (a miss).""" for i, (path, _dist) in enumerate(paths, start=1): @@ -176,7 +198,8 @@ def run_mc3(embedder: _QueryEmbedder, store: VectorStore, *, probes: tuple[CodeProbe, ...] = PROBES, lane_layers: Iterable[str] = LANE_LAYERS, baseline_layers: Iterable[str] = BASELINE_LAYERS, - k: int = _DEFAULT_K, pool: int = _DEFAULT_POOL) -> MC3Result: + k: int = _DEFAULT_K, pool: int = _DEFAULT_POOL, + memberships: MembershipStore | None = None) -> MC3Result: """M-C3: per-probe rank in the code lane vs the docstring-only baseline; the majority-beat verdict (F-CI3). Reproducible: deterministic given (probes, embedder, store, k, pool). A *catastrophic regression* is a probe the baseline finds within top-k but the lane misses @@ -184,8 +207,10 @@ def run_mc3(embedder: _QueryEmbedder, store: VectorStore, *, readings: list[ProbeReading] = [] catastrophic = 0 for p in probes: - lane = ranked_paths(p.query, embedder, store, layers=lane_layers, pool=pool) - base = ranked_paths(p.query, embedder, store, layers=baseline_layers, pool=pool) + lane = ranked_paths(p.query, embedder, store, layers=lane_layers, pool=pool, + memberships=memberships) + base = ranked_paths(p.query, embedder, store, layers=baseline_layers, pool=pool, + memberships=memberships) lr, br = _rank_of(lane, p.answer_paths), _rank_of(base, p.answer_paths) readings.append(ProbeReading(p.probe_id, lr, br)) if br is not None and br <= k and lr is None: diff --git a/tests/integration/test_code_mirror.py b/tests/integration/test_code_mirror.py index 33fce668..1493bb1e 100644 --- a/tests/integration/test_code_mirror.py +++ b/tests/integration/test_code_mirror.py @@ -17,12 +17,13 @@ import hashlib from typing import cast -from core.ingest.code_corpus import code_rows, derive_code_chunks +from core.ingest.code_corpus import code_memberships, code_rows, derive_code_chunks from core.ingest.embed import Embedder from core.ingest.index import _chunk_row, semantic_search from core.kernel.ingest.chunk import Chunk from core.kernel.ingest.pipeline import IngestRecord from core.kernel.provenance import MIRROR_READABLE, Provenance +from core.stores.memberships import MembershipStore from core.stores.vectorstore import VectorStore from eval.code_probes import PROBES from eval.harness.code_retrieval import ( @@ -47,7 +48,10 @@ def _seed_notes(store: VectorStore, emb: HashingEmbedder) -> None: store.add([_chunk_row(rec, c, v) for c, v in zip(rec.chunks, vecs, strict=True)]) -def _seed_code(store: VectorStore, emb: HashingEmbedder) -> None: +def _seed_code(store: VectorStore, memberships: MembershipStore, emb: HashingEmbedder) -> None: + """Land one code file as ATOM rows + its membership fiber (bp-152 D1). Occupancy no longer + rides on the vector row, so the harness resolves paths through the membership join (D3) — the + firewall assertions below are unchanged in kind and are read through the new path.""" src = ( '"""State container."""\n' "def search_nearest_neighbour_embedded_chunks_lancedb(vector):\n" @@ -56,21 +60,24 @@ def _seed_code(store: VectorStore, emb: HashingEmbedder) -> None: chunks = derive_code_chunks("core/store.py", src) vecs = emb.embed_documents([c.text for c in chunks]) blob = hashlib.sha256(src.encode()).hexdigest() - store.add(code_rows("core/store.py", blob, chunks, vecs)) + store.add(code_rows(chunks, vecs, current=True)) + memberships.write_fiber(code_memberships("core/store.py", blob, chunks)) + memberships.reconcile_currency("core/store.py", blob) -def _mixed_store(tmp_path) -> tuple[VectorStore, HashingEmbedder]: +def _mixed_store(tmp_path) -> tuple[VectorStore, MembershipStore, HashingEmbedder]: store = VectorStore(tmp_path / "v.lance", dim=_DIM) + memberships = MembershipStore(tmp_path / "m.sqlite") emb = HashingEmbedder(dim=_DIM) _seed_notes(store, emb) - _seed_code(store, emb) - return store, emb + _seed_code(store, memberships, emb) + return store, memberships, emb def test_default_search_still_surfaces_no_code(tmp_path): """The harness's substrate: the MIRROR_READABLE default never returns code (bp-092 F-CI1, re-checked at the boundary this plan reads from).""" - store, emb = _mixed_store(tmp_path) + store, _memberships, emb = _mixed_store(tmp_path) hits = semantic_search("nearest neighbour embedded chunks lancedb", cast(Embedder, emb), store, k=10) assert hits, "sanity: the authored notes are retrievable" @@ -80,17 +87,18 @@ def test_default_search_still_surfaces_no_code(tmp_path): def test_ranked_paths_returns_only_code_paths(tmp_path): """Both the lane AND the docstring-only baseline read the code lane via provenances={CODE} — never the note mirror. So every returned path is a code source, never a note.""" - store, emb = _mixed_store(tmp_path) + store, memberships, emb = _mixed_store(tmp_path) for layers in (LANE_LAYERS, BASELINE_LAYERS): ranked = ranked_paths("nearest neighbour embedded chunks lancedb", emb, store, - layers=layers, pool=50) + layers=layers, pool=50, memberships=memberships) + assert ranked, "sanity: the code lane is reachable at all through the explicit CODE set" assert all(not p.endswith(".md") for p, _ in ranked) assert all(p == "core/store.py" for p, _ in ranked) def test_mc3_over_a_mixed_store_never_ranks_a_note(tmp_path): - store, emb = _mixed_store(tmp_path) - res = run_mc3(emb, store, probes=PROBES[:3], pool=50) + store, memberships, emb = _mixed_store(tmp_path) + res = run_mc3(emb, store, probes=PROBES[:3], pool=50, memberships=memberships) # the harness completes and every reading is a rank into code paths only (never a note leak) assert len(res.readings) == 3 @@ -99,7 +107,7 @@ def test_mc4_reads_each_class_through_its_own_provenance(tmp_path): """M-C4's cross-space read uses explicit provenance sets: the note side (MIRROR_READABLE) holds no code, the code side holds no note — the deliberate cross-space read never routes through the mirror default (§7).""" - store, _emb = _mixed_store(tmp_path) + store, _memberships, _emb = _mixed_store(tmp_path) code = store.all_rows(provenances={Provenance.CODE}) notes = store.all_rows(provenances=set(MIRROR_READABLE)) assert code and notes diff --git a/tests/integration/test_code_vector_isolation.py b/tests/integration/test_code_vector_isolation.py index aba63fb3..3753fabf 100644 --- a/tests/integration/test_code_vector_isolation.py +++ b/tests/integration/test_code_vector_isolation.py @@ -40,10 +40,13 @@ def _note_rows(store: VectorStore) -> int: def _add_code(store: VectorStore) -> int: + """Land ATOM rows (bp-152 D1): occupancy is shed, `provenance` STAYS. The firewall these tests + guard is a row PREFILTER over `provenance`, so shedding that column would not weaken it — it + would remove it, silently. That is exactly what F-CI1/F-CI5 below still re-check, unchanged.""" src = '"""doc about vectors and dreaming."""\ndef f():\n return 1\n' chunks = derive_code_chunks("m.py", src) vecs = FakeEmbedder().embed_documents([c.text for c in chunks]) - return store.add(code_rows("m.py", "blob1", chunks, vecs)) + return store.add(code_rows(chunks, vecs, current=True)) # ── F-CI1: CODE is unreachable through the mirror surfaces ─────────────────────────────── @@ -119,7 +122,7 @@ def test_layer_migration_preserves_note_rows_bit_identically(tmp_path): store2 = VectorStore(tmp_path / "v.lance", dim=DIM) _add_code(store2) - rows = store2.all_rows() + rows = store2.all_rows(include_atom_rows=True) assert TABLE in store2._db.list_tables().tables notes = {r["id"]: r for r in rows if r["provenance"] == Provenance.AUTHORED_SOLO.value} assert set(notes) == {"n:0", "n:1"} # both note rows survived diff --git a/tests/integration/test_sourceset.py b/tests/integration/test_sourceset.py index 701657d5..17fd5551 100644 --- a/tests/integration/test_sourceset.py +++ b/tests/integration/test_sourceset.py @@ -17,7 +17,7 @@ source_set, source_sets, ) -from core.stores.vectorstore import VectorStore +from core.stores.vectorstore import VectorStore, is_code_atom_row def _row(digest, idx, vec, prov, title="t"): @@ -140,3 +140,62 @@ def test_mixed_provenance_digest_raises(): _row("x", 0, [1.0], Provenance.AUTHORED_SOLO), _row("x", 1, [1.0], Provenance.OBSERVED), ]) + + +# ── the shed-atom guard (bp-152 Item 3, the note's §3 Q5 gap) ── + +def _atom_row(rid, vec): + """A shed CODE-ATOM row (dn-vector-membership-store D1): geometry with NO occupancy — no + `source_path`, no `digest`, because those live in the membership relation now.""" + return {"id": rid, "digest": "", "title": "", "source_path": "", "chunk_index": 0, + "provenance": Provenance.CODE.value, "text": rid, "layer": "code_ast", + "qualname": "", "line_start": 0, "line_end": 0, "vector": vec} + + +def test_shed_code_atom_rows_never_collapse_into_a_bogus_source_set(tmp_path): + """`source_sets(store)` defaults to ALL STRATA by design — "a structural grouping utility, not + a mirror read" — and `group_sources` keys on `digest`. Shed atom rows all carry `digest=''`, so + without a guard every code atom in the corpus collapses into ONE SourceSet keyed `''`. + + It fails SILENTLY: `MixedProvenanceError` needs a digest spanning several provenances and these + rows are uniformly CODE, so a test that merely asserts "no exception" is itself vacuous and is + deliberately not written. The guard lives on the SHED side (`VectorStore.all_rows`) because + `sourceset` is a kernel module that may never learn about the outer-ring membership store (the + C5/D3 ring pin).""" + vs = VectorStore(tmp_path / "v.lance", dim=3) + vs.add([ + _row("a", 0, [1.0, 0.0, 0.0], Provenance.AUTHORED_SOLO), + _row("a", 1, [0.0, 1.0, 0.0], Provenance.AUTHORED_SOLO), + _atom_row("code_ast:aaaa", [0.0, 0.0, 1.0]), + _atom_row("code_ast:bbbb", [0.0, 0.0, 0.5]), + _atom_row("code_text:cccc", [0.5, 0.0, 0.5]), + ]) + # PRECONDITION: the store really holds ≥2 shed code atom rows. Without this the check below + # passes on a store that simply had no code in it. + shed = [r for r in vs.all_rows(provenances={Provenance.CODE}) if is_code_atom_row(r)] + assert len(shed) >= 2 + + # THE FALSIFIER, REPRODUCED: grouping the physical table DOES produce the bogus '' set — so + # the guard below is load-bearing, not a claim about a bug that could not happen. + bogus = {s.digest: s for s in group_sources(vs.all_rows(include_atom_rows=True))} + assert "" in bogus and len(bogus[""]) == len(shed) + + # THE GUARD: the default (all-strata) read excludes them, so no such set is returned. + sets = {s.digest: s for s in source_sets(vs)} + assert "" not in sets + assert set(sets) == {"a"} and len(sets["a"]) == 2 + # ...and the code lane still reads its own rows by asking for them by name (the sync path) + assert len(vs.all_rows(provenances={Provenance.CODE})) == len(shed) + + +def test_the_guard_leaves_a_legacy_per_version_code_row_alone(tmp_path): + """The guard keys on the SHED (`is_code_atom_row`), never on the provenance: a pre-D1 code row + still carries its `(source_path, digest)` and is still a legitimate source object, so it keeps + grouping. Without this, the guard would silently disappear the live store's existing code rows + from every structural consumer before bp-153 has rebuilt anything.""" + vs = VectorStore(tmp_path / "v.lance", dim=3) + legacy = {**_atom_row("m.py:code_ast:dddd", [1.0, 0.0, 0.0]), + "digest": "blob1", "source_path": "m.py", "title": "m.py"} + vs.add([legacy, _atom_row("code_ast:eeee", [0.0, 1.0, 0.0])]) + assert not is_code_atom_row(legacy) # PRECONDITION: it is NOT shed + assert {s.digest for s in source_sets(vs)} == {"blob1"} diff --git a/tests/unit/test_code_corpus.py b/tests/unit/test_code_corpus.py index e084fd60..e4e93edd 100644 --- a/tests/unit/test_code_corpus.py +++ b/tests/unit/test_code_corpus.py @@ -15,6 +15,7 @@ import inspect import subprocess from collections import Counter +from dataclasses import replace from hashlib import sha256 from pathlib import Path @@ -24,16 +25,19 @@ from core.ingest.code_corpus import ( CodeChunk, CodeCorpusSync, + atom_id, code_rows, derive_code_chunks, ) from core.kernel.ingest.chunk import chunk_text from core.kernel.provenance import MIRROR_READABLE, Provenance +from core.stores.memberships import EmbedderIdentity, MembershipStore from core.stores.vectorstore import ( LAYER_CODE_AST, LAYER_CODE_TEXT, LAYER_CODEDOC, VectorStore, + is_code_atom_row, ) from ops.code_snapshot import parse_source from tests.fixtures.embedding import DIM, FakeEmbedder @@ -86,7 +90,8 @@ def test_l0a_headers_name_the_symbol(): assert "import json" in by_qual[""].text assert "# inner comment" in by_qual["Thing.method"].text # fiber coordinates carried on the chunk - assert (by_qual["Thing.method"].line_start, by_qual["Thing.method"].line_end) == (9, 12) + assert (by_qual["Thing.method"].slot_line_start, + by_qual["Thing.method"].slot_line_end) == (9, 12) def test_l0a_oversized_slice_hard_splits_via_chunk_text(): @@ -300,11 +305,61 @@ def test_l0a_oversize_cut_is_canonical_body_scoped(): def test_code_rows_hardcode_code_provenance(): chunks = derive_code_chunks("m.py", _SRC) - rows = code_rows("m.py", "blobsha", chunks, [[0.0] * DIM for _ in chunks]) + rows = code_rows(chunks, [[0.0] * DIM for _ in chunks]) assert {r["provenance"] for r in rows} == {Provenance.CODE.value} assert Provenance.CODE not in MIRROR_READABLE # ∉ the mirror set - assert all(r["digest"] == "blobsha" for r in rows) - # ids are (path, layer, content)-scoped so identical text in two layers stays distinct + # ids are (layer, content)-scoped so identical text in two layers stays distinct + assert len({r["id"] for r in rows}) == len(rows) + + +# ── D1 (bp-152) Item 1: the atom row — path-free identity, occupancy shed, provenance KEPT ── + +def test_the_same_body_at_two_paths_is_one_atom_row_keyed_layer_and_hash(): + """Item 1's acceptance. The same chunk body appearing in two different FILES yields ONE row, + whose id is `f"{layer}:{content_hash}"`. That single character-level change — dropping the path + from the id — IS the atom model: it promotes `code_rows`' old per-path dedup to corpus-wide + dedup (PD-1, owner-ruled in). + + *Falsifier:* two rows appear, i.e. the id still carries the path.""" + body = "def shared_helper(x):\n return x * 2\n" + a = [c for c in derive_code_chunks("pkg/one.py", body) if c.layer == LAYER_CODE_AST] + b = [c for c in derive_code_chunks("pkg/much_deeper/two.py", body) + if c.layer == LAYER_CODE_AST] + # PRECONDITION: the two derivations really are at DIFFERENT paths of different length, and + # each emits the same non-empty symbol set — otherwise "one row" is trivially true. + assert a and {c.qualname for c in a} == {c.qualname for c in b} + assert len("pkg/one.py") != len("pkg/much_deeper/two.py") + + rows_a = code_rows(a, [[0.1] * DIM for _ in a]) + rows_b = code_rows(b, [[0.1] * DIM for _ in b]) + assert {r["id"] for r in rows_a} == {r["id"] for r in rows_b} # ← ONE atom, not two + for r, c in zip(rows_a, a, strict=True): + assert r["id"] == f"{c.layer}:{c.content_hash}" == atom_id(c) + assert "pkg/one.py" not in str(r["id"]) # the path is GONE from the id + + +def test_every_atom_row_keeps_its_provenance_and_sheds_its_occupancy(): + """Item 1's other half, and its falsifier is the worse outcome: `provenance` absent or empty on + an atom row would not weaken the mirror firewall, it would REMOVE it — the firewall is a row + prefilter (`provenance IN (...)`, `prefilter=True`) and holds only if the column is there to + filter. So this asserts presence AND equality to CODE on every row, and asserts the occupancy + columns really did go (otherwise "shed" would be a docstring claim).""" + chunks = derive_code_chunks("m.py", _SRC) + rows = code_rows(chunks, [[0.2] * DIM for _ in chunks]) + assert rows # PRECONDITION: rows exist + for r in rows: + assert r["provenance"] == Provenance.CODE.value # present AND correct + assert (r["source_path"], r["digest"], r["title"]) == ("", "", "") + assert (r["chunk_index"], r["qualname"]) == (0, "") + assert (r["line_start"], r["line_end"]) == (0, 0) + assert is_code_atom_row(r) + # one-layer-one-provenance: an atom never spans strata (the PD-2 fence, as a test invariant) + by_layer: dict[str, set[str]] = {} + for r in rows: + by_layer.setdefault(str(r["layer"]), set()).add(str(r["provenance"])) + assert set(by_layer) == {LAYER_CODE_AST, LAYER_CODE_TEXT, LAYER_CODEDOC} + assert all(v == {Provenance.CODE.value} for v in by_layer.values()) + # ...and identity keeps the layer inside it, so the same text in two layers stays two atoms assert len({r["id"] for r in rows}) == len(rows) @@ -336,6 +391,19 @@ def _git(repo: Path, *args: str) -> str: capture_output=True, text=True).stdout +def _sync(repo: Path, tmp_path: Path, embedder) -> CodeCorpusSync: + """A wired sync driver. `memberships` and `embedder_identity` are REQUIRED fields (bp-152): + the lander cannot record occupancy without the first, and cannot decide reuse honestly without + the second — a reuse across two embedders mixes geometries in one ANN space.""" + return CodeCorpusSync( + repo=repo, + store=VectorStore(tmp_path / "v.lance", dim=DIM), + embedder=embedder, + memberships=MembershipStore(tmp_path / "m.sqlite"), + embedder_identity=EmbedderIdentity(model="fake", dim=DIM), + ) + + @pytest.fixture def repo(tmp_path) -> Path: r = tmp_path / "repo" @@ -351,27 +419,29 @@ def repo(tmp_path) -> Path: def test_seed_then_unchanged_resync_embeds_nothing(repo, tmp_path): - store = VectorStore(tmp_path / "v.lance", dim=DIM) emb = _CountingEmbedder() - sync = CodeCorpusSync(repo=repo, store=store, embedder=emb) + sync = _sync(repo, tmp_path, emb) + store = sync.store seeded = sync.seed() assert seeded.changed_files == 2 and seeded.embedded_rows > 0 after_seed = emb.embedded_texts code_count = len(store.all_rows(provenances={Provenance.CODE})) assert code_count == store.count() # only code rows in this store + fibers = set(sync.memberships.fibers()) + assert {p for p, _ in fibers} == {"a.py", "b.py"} # the D-fiber state is the fiber set now # a second sync with NO change: zero new embeds, store unchanged (the incremental claim) again = sync.sync() assert again.changed_files == 0 and again.embedded_rows == 0 assert emb.embedded_texts == after_seed assert len(store.all_rows(provenances={Provenance.CODE})) == code_count + assert set(sync.memberships.fibers()) == fibers def test_changed_blob_reembeds_only_that_file(repo, tmp_path): - store = VectorStore(tmp_path / "v.lance", dim=DIM) emb = _CountingEmbedder() - sync = CodeCorpusSync(repo=repo, store=store, embedder=emb) + sync = _sync(repo, tmp_path, emb) sync.seed() baseline = emb.embedded_texts @@ -382,73 +452,84 @@ def test_changed_blob_reembeds_only_that_file(repo, tmp_path): report = sync.sync() assert report.changed_files == 1 and report.unchanged_files == 1 assert emb.embedded_texts > baseline # a.py re-embedded - a_rows = [r for r in store.all_rows(provenances={Provenance.CODE}) - if r["source_path"] == "a.py"] - assert any("changed" in str(r["text"]) for r in a_rows) + # [banner: correction] the changed text is found through the MEMBERSHIP fiber now — the atom + # row carries no `source_path` to filter on (D1). Same claim, resolved through the join (D3). + head = _git(repo, "rev-parse", "HEAD:a.py").strip() + by_id = {str(r["id"]): r for r in sync.store.all_rows(provenances={Provenance.CODE})} + a_texts = [str(by_id[m.content_id]["text"]) for m in sync.memberships.fiber("a.py", head)] + assert any("changed" in t for t in a_texts) def test_vanished_file_is_retained_but_marked_superseded(repo, tmp_path): """Keep-and-link (dn-temporal-code-corpus D2, bp-099 — reverses the old delete): a vanished file's rows are RETAINED (never deleted) but flipped current=false, so the current-view no longer surfaces them while history is preserved.""" - store = VectorStore(tmp_path / "v.lance", dim=DIM) - sync = CodeCorpusSync(repo=repo, store=store, embedder=FakeEmbedder()) + sync = _sync(repo, tmp_path, FakeEmbedder()) sync.seed() + b_head = _git(repo, "rev-parse", "HEAD:b.py").strip() + v_before = len(sync.store.atom_rows()) + assert sync.memberships.fiber("b.py", b_head) # PRECONDITION: b.py really landed (repo / "b.py").unlink() _git(repo, "add", "-A") _git(repo, "commit", "-qm", "drop b") report = sync.sync() assert report.deleted_files == 1 - assert report.superseded_rows > 0 # b.py's rows flipped, not deleted - all_code = store.all_rows(provenances={Provenance.CODE}) - # b.py is RETAINED (the falsifier: a superseded row must never be deleted) - assert {r["source_path"] for r in all_code} == {"a.py", "b.py"} - b_rows = [r for r in all_code if r["source_path"] == "b.py"] - assert b_rows and all(r["current"] is False for r in b_rows) - a_rows = [r for r in all_code if r["source_path"] == "a.py"] - assert all(r["current"] is True for r in a_rows) + assert report.superseded_rows > 0 # b.py's OCCUPANCIES flipped, not deleted + # b.py is RETAINED (the falsifier: a superseded occupancy must never be deleted), and the + # ATOMS are untouched — append-only means a vanished file removes no geometry at all. + b_fiber = sync.memberships.fiber("b.py", b_head) + assert b_fiber and not any(m.current for m in b_fiber) + assert not any(m.tombstoned for m in b_fiber) # superseded is not purged + a_head = _git(repo, "rev-parse", "HEAD:a.py").strip() + assert all(m.current for m in sync.memberships.fiber("a.py", a_head)) + assert len(sync.store.atom_rows()) == v_before # |V| never decreases (§4) def test_changed_blob_keeps_and_links_old_version(repo, tmp_path): """D2: on a changed blob the OLD version survives current=false (same ids, vectors intact) and the NEW version lands current=true. The falsifier: any superseded row deleted.""" - store = VectorStore(tmp_path / "v.lance", dim=DIM) - sync = CodeCorpusSync(repo=repo, store=store, embedder=FakeEmbedder()) + sync = _sync(repo, tmp_path, FakeEmbedder()) sync.seed() - before = {(r["id"], r["digest"], tuple(r["vector"])) - for r in store.all_rows(provenances={Provenance.CODE}) - if r["source_path"] == "a.py"} + old_blob = _git(repo, "rev-parse", "HEAD:a.py").strip() + old_fiber = sync.memberships.fiber("a.py", old_blob) + before = {(str(r["id"]), tuple(r["vector"])) for r in sync.store.atom_rows()} + assert old_fiber and before # PRECONDITION: v1 really landed (repo / "a.py").write_text("def a():\n return 1 + 1 # changed\n") _git(repo, "add", "-A") _git(repo, "commit", "-qm", "two") report = sync.sync() assert report.superseded_rows > 0 + new_blob = _git(repo, "rev-parse", "HEAD:a.py").strip() + assert new_blob != old_blob - a_rows = [r for r in store.all_rows(provenances={Provenance.CODE}) - if r["source_path"] == "a.py"] - old = [r for r in a_rows if r["current"] is False] - new = [r for r in a_rows if r["current"] is True] - assert old and new # BOTH versions retained - assert any("changed" in str(r["text"]) for r in new) # the new version is current - assert not any("changed" in str(r["text"]) for r in old) - # the old rows survive with the SAME ids/digest/vectors (vectors carried through the flip) - after_old = {(r["id"], r["digest"], tuple(r["vector"])) for r in old} - assert before <= after_old + # BOTH versions retained as fibers — the old one superseded, never deleted + now_old = sync.memberships.fiber("a.py", old_blob) + now_new = sync.memberships.fiber("a.py", new_blob) + assert now_old == [replace(m, current=False) for m in old_fiber] # same rows, flag flipped + assert now_new and all(m.current for m in now_new) + by_id = {str(r["id"]): r for r in sync.store.atom_rows()} + assert any("changed" in str(by_id[m.content_id]["text"]) for m in now_new) + assert not any("changed" in str(by_id[m.content_id]["text"]) for m in now_old) + # every atom that existed before survives with its vector intact (append-only, section 4) + assert before <= {(str(r["id"]), tuple(r["vector"])) for r in sync.store.atom_rows()} def test_default_search_is_current_view_history_is_opt_in(repo, tmp_path): """D3: a superseded version never surfaces on the default search; include_superseded=True returns it. A deterministic fake embedder + a real temp lance store (no Ollama).""" - store = VectorStore(tmp_path / "v.lance", dim=DIM) emb = FakeEmbedder() - sync = CodeCorpusSync(repo=repo, store=store, embedder=emb) + sync = _sync(repo, tmp_path, emb) + store = sync.store sync.seed() (repo / "a.py").write_text("def a():\n return 42 # newtoken\n") _git(repo, "add", "-A") _git(repo, "commit", "-qm", "two") sync.sync() + # PRECONDITION (the named degenerate input): an atom whose every occupancy is superseded must + # EXIST, or the current-filter below passes vacuously. + assert any(not bool(r["current"]) for r in store.atom_rows()) q = emb.embed_documents(["def a"])[0] current_only = store.search(q, k=50, provenances={Provenance.CODE}) assert current_only and all(r["current"] is True for r in current_only) @@ -467,15 +548,19 @@ def test_current_column_additive_migration_preserves_rows(tmp_path): from core.stores.vectorstore import TABLE path = tmp_path / "v.lance" chunks = derive_code_chunks("m.py", _SRC) - landed = code_rows("m.py", "blob0", chunks, [[0.1] * DIM for _ in chunks]) + # a LEGACY (pre-D1) code row still carries its occupancy columns — that IS the shape being + # migrated, and restoring it here is what lets the migration assertion mean anything. + landed = [{**r, "id": f"m.py:{r['id']}", "source_path": "m.py", "digest": "blob0", + "title": "m.py"} + for r in code_rows(chunks, [[0.1] * DIM for _ in chunks])] legacy = [{k: v for k, v in r.items() if k != "current"} for r in landed] # strip current raw = lancedb.connect(str(path)) raw.create_table(TABLE, data=legacy) # a pre-bp-099 (no-current) table store = VectorStore(path, dim=DIM) # opens the legacy table - store.add(code_rows("n.py", "blob1", derive_code_chunks("n.py", _SRC), - [[0.2] * DIM for _ in derive_code_chunks("n.py", _SRC)])) + n_chunks = derive_code_chunks("n.py", _SRC) + store.add(code_rows(n_chunks, [[0.2] * DIM for _ in n_chunks])) rows = store.all_rows(provenances={Provenance.CODE}) assert all("current" in r for r in rows) m_rows = [r for r in rows if r["source_path"] == "m.py"] diff --git a/tests/unit/test_code_lineage.py b/tests/unit/test_code_lineage.py index c02a5592..80ba84e5 100644 --- a/tests/unit/test_code_lineage.py +++ b/tests/unit/test_code_lineage.py @@ -16,8 +16,8 @@ import pytest from core.ingest.code_corpus import CodeCorpusSync -from core.kernel.provenance import Provenance from core.kernel.temporal.boundary import delta_D_squared_is_zero, poset_from_chains +from core.stores.memberships import EmbedderIdentity, MembershipStore from core.stores.vectorstore import VectorStore from ops.code_lineage import ( capture_commit_diffs, @@ -154,43 +154,57 @@ def _first_commit(repo: Path) -> str: # ── the history backfill: current flags · idempotent · parse-fail counted ────────────────── +def _sync(repo, tmp_path) -> CodeCorpusSync: + return CodeCorpusSync( + repo=repo, store=VectorStore(tmp_path / "v.lance", dim=DIM), embedder=FakeEmbedder(), + memberships=MembershipStore(tmp_path / "m.sqlite"), + embedder_identity=EmbedderIdentity(model="fake", dim=DIM)) + + def test_backfill_embeds_history_with_current_flags(ledger, repo, tmp_path): - store = VectorStore(tmp_path / "v.lance", dim=DIM) - sync = CodeCorpusSync(repo=repo, store=store, embedder=FakeEmbedder()) + """[banner: correction] The per-version state re-homes from the vector row's + `(source_path, digest)` to the MEMBERSHIP fiber (bp-152 D1): an atom row carries neither + column, and a version IS its fiber. Same three claims — three versions land, exactly HEAD is + current, a re-run embeds nothing — read at their new home.""" + sync = _sync(repo, tmp_path) report = sync.backfill(ledger_versions(ledger)) assert report.embedded_rows > 0 assert report.parse_failures >= 1 # broken.py counted, still embedded - code = store.all_rows(provenances={Provenance.CODE}) - # f.py: three versions embedded; exactly the HEAD blob is current=true, the older two false - f_by_digest = {r["digest"]: r["current"] for r in code if r["source_path"] == "f.py"} - assert len(f_by_digest) == 3 + m = sync.memberships + # f.py: three versions land as three fibers; exactly the HEAD blob's is current + f_blobs = m.blobs_of("f.py") + assert len(f_blobs) == 3 head_f = _git(repo, "rev-parse", "HEAD:f.py").strip() - assert f_by_digest[head_f] is True - assert sum(1 for c in f_by_digest.values() if c is True) == 1 - assert sum(1 for c in f_by_digest.values() if c is False) == 2 + current = [b for b in f_blobs if all(x.current for x in m.fiber("f.py", b))] + assert current == [head_f] + assert all(not any(x.current for x in m.fiber("f.py", b)) + for b in f_blobs if b != head_f) # broken.py embedded despite the parse failure (L0b windows + module shell) - assert any(r["source_path"] == "broken.py" for r in code) + assert m.blobs_of("broken.py") - # idempotent: a re-run embeds nothing + # idempotent: a re-run embeds nothing and adds no occupancy + m_before = m.count() again = sync.backfill(ledger_versions(ledger)) - assert again.embedded_rows == 0 + assert again.embedded_rows == 0 and again.membership_rows == 0 + assert m.count() == m_before def test_composed_supersession_edge_resolves_to_embedded_nodes(ledger, repo, tmp_path): """D5, the realized edge: a `commit_diffs` modify row's old_blob AND new_blob both resolve to embedded rows in the backfilled store — the integrator's landing surface.""" - store = VectorStore(tmp_path / "v.lance", dim=DIM) - sync = CodeCorpusSync(repo=repo, store=store, embedder=FakeEmbedder()) + sync = _sync(repo, tmp_path) sync.backfill(ledger_versions(ledger)) capture_commit_diffs(ledger, repo, ledger_commits(ledger)) - embedded_digests = {str(r["digest"]) - for r in store.all_rows(provenances={Provenance.CODE})} + # [banner: correction] The note's section 6 re-home (2), F6: an endpoint is resolvable IFF its + # membership fiber is non-empty. "By digest in the vector store" cannot be asked of an atom + # row — the column left it — and the fiber is the same fact at a sturdier home. modify = ledger.execute( "SELECT path, old_blob, new_blob FROM commit_diffs " "WHERE old_blob != '' AND new_blob != '' LIMIT 1").fetchone() assert modify is not None - _path, old_blob, new_blob = modify - assert old_blob in embedded_digests and new_blob in embedded_digests + path, old_blob, new_blob = modify + assert sync.memberships.fiber(path, old_blob) + assert sync.memberships.fiber(path, new_blob) diff --git a/tests/unit/test_code_retrieval.py b/tests/unit/test_code_retrieval.py index 437ce0f2..4ec1f132 100644 --- a/tests/unit/test_code_retrieval.py +++ b/tests/unit/test_code_retrieval.py @@ -13,8 +13,9 @@ import hashlib from pathlib import Path -from core.ingest.code_corpus import code_rows, derive_code_chunks +from core.ingest.code_corpus import code_memberships, code_rows, derive_code_chunks from core.kernel.provenance import Provenance +from core.stores.memberships import MembershipStore from core.stores.vectorstore import VectorStore from eval.code_probes import PROBES, CodeProbe, probe_set_hash from eval.harness.code_retrieval import ( @@ -63,12 +64,19 @@ def test_probe_set_hash_is_stable_and_order_independent(): # ── M-C3: retrieval — code lane vs docstring-only baseline ──────────────────────────────── -def _seed_code(store: VectorStore, sources: dict[str, str], embedder: HashingEmbedder) -> None: +def _seed_code(store: VectorStore, memberships: MembershipStore, sources: dict[str, str], + embedder: HashingEmbedder) -> None: + """[banner: correction] Seeds ATOM rows plus their membership fibers (bp-152 D1). The atom row + carries no `source_path`, so the M-C3 ranking resolves paths through the membership join (D3) + — which is also why a SHARED atom would now rank every file it occupies, the intended reading + of one idea living in two places.""" for path, src in sources.items(): chunks = derive_code_chunks(path, src) vecs = embedder.embed_documents([c.text for c in chunks]) blob = hashlib.sha256(src.encode()).hexdigest() - store.add(code_rows(path, blob, chunks, vecs)) + store.add(code_rows(chunks, vecs, current=True)) + memberships.write_fiber(code_memberships(path, blob, chunks)) + memberships.reconcile_currency(path, blob) # The query terms live ONLY in identifiers/bodies (not docstrings/comments), so the @@ -172,9 +180,10 @@ def test_mc3_no_signal_when_lane_equals_baseline(tmp_path): """F-CI3 path: with the lane restricted to the baseline layers every probe ties, so the majority-beat fails and the verdict is NO_SIGNAL — a result, not a crash (still seals).""" store = VectorStore(tmp_path / "v.lance", dim=_DIM) + memberships = MembershipStore(tmp_path / "m.sqlite") emb = HashingEmbedder(dim=_DIM) - _seed_code(store, _SOURCES, emb) - res = run_mc3(emb, store, probes=_PROBES3, pool=50, + _seed_code(store, memberships, _SOURCES, emb) + res = run_mc3(emb, store, probes=_PROBES3, pool=50, memberships=memberships, lane_layers=BASELINE_LAYERS, baseline_layers=BASELINE_LAYERS) assert res.lane_wins == 0 and res.baseline_wins == 0 assert res.verdict is MC3Verdict.NO_SIGNAL @@ -182,13 +191,33 @@ def test_mc3_no_signal_when_lane_equals_baseline(tmp_path): def test_ranked_paths_dedupes_by_source_and_filters_layer(tmp_path): store = VectorStore(tmp_path / "v.lance", dim=_DIM) + memberships = MembershipStore(tmp_path / "m.sqlite") emb = HashingEmbedder(dim=_DIM) - _seed_code(store, _SOURCES, emb) + _seed_code(store, memberships, _SOURCES, emb) + # PRECONDITION: the rows really are shed atom rows, so this exercises the D3 JOIN and not a + # leftover `source_path` on the row — without it the test would pass on the pre-split shape. + assert all(not str(r["source_path"]) for r in store.atom_rows()) lane = ranked_paths("nearest neighbour embedded chunks lancedb", emb, store, - layers=LANE_LAYERS, pool=50) + layers=LANE_LAYERS, pool=50, memberships=memberships) paths = [p for p, _ in lane] - assert paths[0] == "core/store.py" + assert paths and paths[0] == "core/store.py" assert len(paths) == len(set(paths)), "one entry per source path" + assert set(paths) <= set(_SOURCES) + + +def test_ranked_paths_without_memberships_resolves_no_path_for_an_atom_row(tmp_path): + """The join is not optional decoration: an atom row has no occupancy ON it, so a caller that + does not hand in the membership store gets nothing back rather than a silently wrong path. + Failing empty is the honest shape — a shed row simply does not know where it lives.""" + store = VectorStore(tmp_path / "v.lance", dim=_DIM) + memberships = MembershipStore(tmp_path / "m.sqlite") + emb = HashingEmbedder(dim=_DIM) + _seed_code(store, memberships, _SOURCES, emb) + with_join = ranked_paths("nearest neighbour embedded chunks lancedb", emb, store, + layers=LANE_LAYERS, pool=50, memberships=memberships) + assert with_join # PRECONDITION: the query DOES retrieve + assert ranked_paths("nearest neighbour embedded chunks lancedb", emb, store, + layers=LANE_LAYERS, pool=50) == [] # ── M-C4: cross-space geometry — informative vs degenerate ──────────────────────────────── diff --git a/tests/unit/test_memberships.py b/tests/unit/test_memberships.py new file mode 100644 index 00000000..7ce4dcc6 --- /dev/null +++ b/tests/unit/test_memberships.py @@ -0,0 +1,687 @@ +"""core/stores/memberships.py + the D2 lander — the membership store (bp-152). + +Every criterion here names the **degenerate input** on which it would pass without testing its +claim, and asserts the precondition that rules it out FIRST (the false-success rule, owner-agreed). +Four separate defects in this wave were "a test that would pass without testing its claim", so the +preconditions are not decoration: + + * §8(a) revert — a do-nothing lander also embeds zero, so the CURRENCY assertions carry the claim, + and a short-circuiting lander is exhibited beside them to show they have teeth. + * §8(b) fork — a store that never dedups makes "the other fiber untouched" vacuous, so + `|V| < Σ chunks` and "the shared atom has 2 current memberships" are asserted first. + * §8(c) purge — a purge that silently no-ops also "never deletes", so the RECORDED HOLE is what + is asserted. + * §8(d) retrieval — an all-current fixture passes any current-filter, so a superseded occupancy + is asserted to exist first. + * §8(e) crash — a "crash" injected after all writes makes repair vacuous, so the injection point + is asserted (|V| grew, |M| did not) and the orphan is asserted OBSERVABLE. + * §8(f) invariants — each reddens under a seeded violation; a fixture missing any of the four + shapes (revert, side-branch, shared atom, duplicate L0b pair) makes its invariant vacuous. + * Item 2's A2 coordinates — a leaf-symbol fixture makes the subset assertion vacuous (span == + coverage exactly), so the fixture carries a class with methods AND a module shell. + +Deterministic throughout: a fake embedder, a temp lance store, a temp SQLite membership store. No +Ollama, no network. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path + +import pytest + +from core.ingest.code_corpus import ( + CodeChunk, + CodeLander, + atom_id, + code_memberships, + code_rows, + derive_code_chunks, +) +from core.kernel.provenance import Provenance +from core.stores.memberships import ( + EmbedderIdentity, + Membership, + MembershipStore, + current_any_drift, + purge_atom, + repair_current_any, + resolve_occupancies, +) +from core.stores.vectorstore import ( + LAYER_CODE_AST, + LAYER_CODE_TEXT, + VectorStore, +) +from ops.code_snapshot import parse_source +from tests.fixtures.embedding import DIM, FakeEmbedder + +_REPO_ROOT = Path(__file__).resolve().parents[2] +_EMB = EmbedderIdentity(model="fake-embedder", dim=DIM) +_EMB_OTHER = EmbedderIdentity(model="a-different-model", dim=DIM) + + +class _CountingEmbedder(FakeEmbedder): + """Counts embed calls — the only way to see "reuse" at all: identical text gives an identical + vector whether it was recomputed or reused, so the vector alone can never show it.""" + + def __init__(self) -> None: + self.embedded: list[str] = [] + + def embed_documents(self, texts: list[str]) -> list[list[float]]: + self.embedded.extend(texts) + return super().embed_documents(texts) + + +@dataclass +class Bench: + vectors: VectorStore + memberships: MembershipStore + embedder: _CountingEmbedder + lander: CodeLander + + def n_vectors(self) -> int: + """|V| — atom rows only.""" + return len(self.vectors.atom_rows()) + + +@pytest.fixture +def bench(tmp_path: Path) -> Bench: + vectors = VectorStore(tmp_path / "v.lance", dim=DIM) + memberships = MembershipStore(tmp_path / "m.sqlite") + embedder = _CountingEmbedder() + return Bench(vectors=vectors, memberships=memberships, embedder=embedder, + lander=CodeLander(vectors=vectors, memberships=memberships, + embedder=embedder, embedder_identity=_EMB)) + + +def _chunk(qualname: str, body: str, *, layer: str = LAYER_CODE_AST, + span: tuple[int, int] = (1, 1)) -> CodeChunk: + """A hand-built chunk. Used where the ACCEPTANCE needs exact atom arithmetic ("exactly 1 new + atom"): a real derivation of a real file also recuts its single L0b window on any edit, which + is correct behavior and useless for counting.""" + return CodeChunk(layer=layer, qualname=qualname, slot_line_start=span[0], + slot_line_end=span[1], text=f"# x:{qualname}\n{body}", canonical_body=body) + + +# ── Item 2 — the fiber round-trip, and the A2 coordinate reading ───────────────────────── + +_A2_SRC = ( + '"""Module doc."""\n' # 1 ← module shell + "import os\n" # 2 ← module shell + "\n" # 3 ← module shell + "\n" # 4 ← module shell + "class Foo:\n" # 5 ← Foo + ' """Foo doc."""\n' # 6 ← Foo + "\n" # 7 ← Foo + " LIMIT = 3\n" # 8 ← Foo + "\n" # 9 ← Foo + " def bar(self):\n" # 10 ← Foo.bar + " return 1\n" # 11 ← Foo.bar + "\n" # 12 ← Foo (blank lines fall to the enclosing owner) + " # a stray comment\n" # 13 ← Foo + " def baz(self):\n" # 14 ← Foo.baz + " return 2\n" # 15 ← Foo.baz + "\n" # 16 ← module shell + "\n" # 17 ← module shell + "def top():\n" # 18 ← top (a LEAF: span == coverage, the control) + " return Foo()\n" # 19 ← top +) + + +def test_fiber_round_trips_the_versions_chunks_in_derivation_order(bench: Bench) -> None: + """Item 2: a fiber written and read back IS the version's chunk list, in `chunk_index` order, + and Σ fiber sizes = |M| on the fixture.""" + chunks = derive_code_chunks("pkg/a.py", _A2_SRC) + assert len(chunks) > 3 # PRECONDITION: a real derivation + bench.memberships.write_fiber(code_memberships("pkg/a.py", "blobA", chunks)) + + fiber = bench.memberships.fiber("pkg/a.py", "blobA") + assert [m.chunk_index for m in fiber] == list(range(len(chunks))) + assert [m.content_id for m in fiber] == [atom_id(c) for c in chunks] + assert [m.layer for m in fiber] == [c.layer for c in chunks] + assert sum(bench.memberships.fiber_sizes().values()) == bench.memberships.count() + + +def test_slot_coordinates_are_the_declared_extent_not_the_atoms_text_coverage( + bench: Bench) -> None: + """Item 2 / Amendment A2 / issue #34 — the assertion that makes the RENAME more than cosmetic. + + `slot_line_start`/`slot_line_end` are the SLOT's declared extent, never the atom's text + coverage. Every leaf-symbol fixture passes either way (there the two coincide exactly), which + is precisely how this went unnoticed — so the fixture below carries a class WITH METHODS and a + non-empty MODULE SHELL, and the precondition asserts it does.""" + path = "pkg/a.py" + src_lines = _A2_SRC.splitlines() + shape = parse_source(path, "", _A2_SRC) + by_qual = {s.qualname: s for s in shape.symbols} + + # PRECONDITION (the named degenerate input): a NESTED symbol and a non-empty module shell. + # A leaf-only fixture makes every "strict subset" below vacuous. + assert "Foo" in by_qual and "Foo.bar" in by_qual and "Foo.baz" in by_qual + assert by_qual["Foo"].lineno < by_qual["Foo.bar"].lineno <= by_qual["Foo"].end_lineno + + chunks = derive_code_chunks(path, _A2_SRC) + bench.memberships.write_fiber(code_memberships(path, "blobA", chunks)) + l0a = {m.slot: m for m in bench.memberships.fiber(path, "blobA") + if m.layer == LAYER_CODE_AST} + bodies = {c.qualname: c.canonical_body for c in chunks if c.layer == LAYER_CODE_AST} + assert set(l0a) >= {"", "Foo", "Foo.bar", "top"} # PRECONDITION: the shell IS emitted + assert bodies[""].strip() # ...and is NOT empty + + def span_lines(m: Membership) -> list[str]: + return src_lines[m.slot_line_start - 1:m.slot_line_end] + + # (1) the span is the SYMBOL's declared extent — `owner.lineno, owner.end_lineno`, verbatim + for q in ("Foo", "Foo.bar", "Foo.baz", "top"): + assert (l0a[q].slot_line_start, l0a[q].slot_line_end) == (by_qual[q].lineno, + by_qual[q].end_lineno) + + # (2) the atom's text coverage is a STRICT SUBSET of that span for the nested case: `Foo`'s + # chunk holds the class statement, its docstring, `LIMIT` and the stray comment — but NOT + # `bar`, NOT `baz`, which carved their lines out of the parent (innermost-owner partition). + foo_body = set(bodies["Foo"].splitlines()) + foo_span = set(span_lines(l0a["Foo"])) + assert foo_body < foo_span # ← strict subset, the A2 claim + assert " def bar(self):" in foo_span # ...in the SPAN + assert " def bar(self):" not in foo_body # ...never in the TEXT + assert " LIMIT = 3" in foo_body + + # (3) the module shell is the maximal case: its extent is the ENTIRE FILE by construction, + # for four lines of preamble. Rendering the range as "the atom" renders the whole file. + shell = l0a[""] + assert (shell.slot_line_start, shell.slot_line_end) == (1, len(src_lines)) + shell_body = set(bodies[""].splitlines()) + assert shell_body < set(src_lines) + assert len(shell_body) < len(src_lines) / 2 # the divergence is not marginal + + # (4) the CONTROL that makes (2)/(3) meaningful: for a leaf symbol they coincide exactly, so a + # leaf-only fixture would have proved nothing at all. + assert set(bodies["top"].splitlines()) == set(span_lines(l0a["top"])) + + +def test_a_blob_with_two_identical_windows_stores_two_membership_rows(bench: Bench) -> None: + """Item 2's falsifier: the atom side dedups, the membership side must NOT. + + `code_rows` collapses duplicates via `by_id.setdefault` — correct for geometry, and WRONG for + occupancy. Two byte-identical L0b windows in one blob are TWO occupancies with distinct + `chunk_index` (the F5 multiset pin); inheriting the collapse would make one of them vanish by + key collision with nothing to see.""" + w = _chunk("", "x = 1\ny = 2\n", layer=LAYER_CODE_TEXT, span=(1, 2)) + chunks = [w, w] + # PRECONDITION: the two windows really are the same atom — otherwise "two rows" is trivial. + assert atom_id(chunks[0]) == atom_id(chunks[1]) + + bench.vectors.add(code_rows(chunks, [[0.5] * DIM, [0.5] * DIM])) + bench.memberships.write_fiber(code_memberships("m.py", "blob1", chunks)) + + assert len(bench.vectors.atom_rows()) == 1 # the ATOM side dedups (one geometry) + fiber = bench.memberships.fiber("m.py", "blob1") + assert len(fiber) == 2 # ← the MEMBERSHIP side does not + assert [m.chunk_index for m in fiber] == [0, 1] + assert {m.content_id for m in fiber} == {atom_id(w)} + assert bench.memberships.n_occ(atom_id(w), current_only=False) == 2 # multiset reading + assert bench.memberships.n_doc(atom_id(w), current_only=False) == 1 # document reading + + +def test_the_membership_store_is_never_imported_by_the_kernel() -> None: + """The C5/D3 ring pin, structurally. A `core/kernel/**` import of this module would mechanically + demote `sourceset` from the inner-ring fixed point; memberships enter the kernel as DATA, + through the existing `RowSource` protocol. The negative control asserts the scanner works.""" + kernel = sorted((_REPO_ROOT / "core" / "kernel").rglob("*.py")) + assert len(kernel) > 20 # PRECONDITION: the scan sees the tree + offenders = [p for p in kernel if "core.stores.memberships" in p.read_text(encoding="utf-8")] + assert offenders == [] + # negative control: the scanner DOES fire where the import exists (the lander is an outer-ring + # module and imports it) — so the emptiness above is a fact, not a broken check. + assert "core.stores.memberships" in ( + _REPO_ROOT / "core" / "ingest" / "code_corpus.py").read_text(encoding="utf-8") + + +# ── Item 4 / §8(a) — the C1 case: land A → B → A ───────────────────────────────────────── + +_SRC_A = '"""Doc A."""\n\n\ndef work(x):\n return x + 1\n' +_SRC_B = '"""Doc A."""\n\n\ndef work(x):\n return x + 2 # edited\n' + + +def test_revert_relands_zero_and_currency_converges(bench: Bench) -> None: + """§8(a), the C1 case — the reason this plan exists. + + Land A → B → A (a real revert: byte-identical bytes return under a new landing). Zero vector + inserts, zero new membership rows, AND currency converges: A's fiber current, B's not. + + *Degenerate input:* a do-nothing lander ALSO lands zero vectors and zero rows — so the two + currency assertions are the ones carrying the claim, and the mutation check below exhibits the + lander that satisfies the counts and fails them.""" + path = "pkg/w.py" + ch_a = derive_code_chunks(path, _SRC_A) + ch_b = derive_code_chunks(path, _SRC_B) + + first = bench.lander.land(path, "blobA", ch_a) + assert first.atoms_embedded > 0 # PRECONDITION: A's atoms really exist + v_after_a = bench.n_vectors() + + bench.lander.land(path, "blobB", ch_b) + assert bench.n_vectors() > v_after_a # PRECONDITION: B really differs from A + v_after_b = bench.n_vectors() + assert all(m.current for m in bench.memberships.fiber(path, "blobB")) + + embeds_before = len(bench.embedder.embedded) + third = bench.lander.land(path, "blobA", ch_a) # ← the revert + + # the reuse half + assert third.atoms_embedded == 0 + assert third.membership_rows == 0 # the fiber already stood + assert len(bench.embedder.embedded) == embeds_before # nothing re-embedded + assert bench.n_vectors() == v_after_b # |V| did not move (append-only) + + # the CLAIM: currency converged — this is what a short-circuiting lander gets wrong + assert all(m.current for m in bench.memberships.fiber(path, "blobA")) + assert not any(m.current for m in bench.memberships.fiber(path, "blobB")) + assert third.currency.made_current > 0 and third.currency.superseded > 0 + assert current_any_drift(bench.vectors, bench.memberships) == [] + + +def test_a_lander_that_short_circuits_on_an_existing_fiber_leaves_b_current( + bench: Bench) -> None: + """The MUTATION CHECK for the test above: the "existing fiber ⇒ no-op" lander — the design's + own first draft — satisfies every count assertion (zero embeds, zero new rows) and leaves the + store claiming **B** is HEAD. Without this, "currency converged" could be a claim about a + property nothing could ever violate.""" + path = "pkg/w.py" + bench.lander.land(path, "blobA", derive_code_chunks(path, _SRC_A)) + bench.lander.land(path, "blobB", derive_code_chunks(path, _SRC_B)) + + # the short-circuit, spelled out: steps 1-3 only, step 4 skipped because step 3 was a no-op + chunks = derive_code_chunks(path, _SRC_A) + written = bench.memberships.write_fiber(code_memberships(path, "blobA", chunks)) + assert written == 0 # "nothing to do" — the tempting reading + + assert not any(m.current for m in bench.memberships.fiber(path, "blobA")) + assert all(m.current for m in bench.memberships.fiber(path, "blobB")) # ← silent corruption + + # ...and the real lander repairs it by convergence, from exactly this state + bench.lander.land(path, "blobA", chunks) + assert all(m.current for m in bench.memberships.fiber(path, "blobA")) + assert not any(m.current for m in bench.memberships.fiber(path, "blobB")) + + +def test_reuse_is_invalidated_by_a_change_of_embedder(tmp_path: Path) -> None: + """The embedder pin (owner confirmation 2026-08-01). Atom presence is keyed to + `(layer, content_hash)` AND the embedder identity, so a model change re-embeds instead of + serving a vector from another geometry into the same ANN space. + + *Degenerate input:* a suite that only ever exercises one embedder cannot see this bug at all — + hence a case of its own, with the same-embedder control asserted first so "re-embedded" is + known to mean "the embedder changed" and not "reuse never worked".""" + vectors = VectorStore(tmp_path / "v.lance", dim=DIM) + memberships = MembershipStore(tmp_path / "m.sqlite") + embedder = _CountingEmbedder() + chunks = derive_code_chunks("pkg/w.py", _SRC_A) + + same = CodeLander(vectors=vectors, memberships=memberships, embedder=embedder, + embedder_identity=_EMB) + first = same.land("pkg/w.py", "blobA", chunks) + assert first.atoms_embedded > 0 + # CONTROL: under the SAME embedder, a re-land reuses everything + assert same.land("pkg/w.py", "blobA", chunks).atoms_embedded == 0 + + changed = CodeLander(vectors=vectors, memberships=memberships, embedder=embedder, + embedder_identity=_EMB_OTHER) + after = changed.land("pkg/w.py", "blobA", chunks) + assert after.atoms_embedded == first.atoms_embedded # ← every reuse invalidated + assert after.atoms_reused == 0 + + +# ── Item 5 / §8(b) — fork semantics ────────────────────────────────────────────────────── + +def _fork_bench(bench: Bench) -> tuple[CodeChunk, CodeChunk, CodeChunk]: + """Two files sharing one atom. Hand-built so "exactly 1 new atom" is exact arithmetic.""" + shared = _chunk("shared", "def shared():\n return 7\n", span=(1, 2)) + only_x = _chunk("only_x", "def only_x():\n return 1\n", span=(4, 5)) + only_y = _chunk("only_y", "def only_y():\n return 2\n", span=(4, 5)) + bench.lander.land("x.py", "x1", [shared, only_x]) + bench.lander.land("y.py", "y1", [shared, only_y]) + return shared, only_x, only_y + + +def test_fork_one_atom_two_memberships_and_an_edit_touches_only_one_lineage( + bench: Bench) -> None: + """§8(b). Preconditions FIRST: the shared atom has 2 current memberships and |V| < Σ chunks. + + *Degenerate input:* a store that never dedups makes "the other file's fiber untouched" vacuous + — it would be untouched because nothing was ever shared. The precondition reddens on it.""" + shared, only_x, only_y = _fork_bench(bench) + shared_id = atom_id(shared) + sigma_chunks = bench.memberships.count() # 2 chunks landed per file, 2 files + + # PRECONDITIONS — without these the whole test is about a store that never shared anything + assert sigma_chunks == 4 + assert bench.n_vectors() == 3 < sigma_chunks # ← |V| < Σ chunks: the dedup happened + assert bench.memberships.n_doc(shared_id) == 2 # the shared atom has 2 current homes + assert len(bench.memberships.occupancies(shared_id)) == 2 + + x_before = bench.memberships.fiber("x.py", "x1") + embeds_before = len(bench.embedder.embedded) + + # edit ONE file: the same slot, new content + edited = _chunk("only_y", "def only_y():\n return 99\n", span=(4, 5)) + report = bench.lander.land("y.py", "y2", [shared, edited]) + + assert report.atoms_embedded == 1 # ← exactly 1 new atom + assert len(bench.embedder.embedded) == embeds_before + 1 + assert bench.n_vectors() == 4 + + # 1 membership swap: y's CURRENT occupancy set differs from before in exactly one atom + now = {m.content_id for m in bench.memberships.fiber("y.py", "y2")} + was = {m.content_id for m in bench.memberships.fiber("y.py", "y1")} + assert now - was == {atom_id(edited)} # one in ... + assert was - now == {atom_id(only_y)} # ... one out + assert shared_id in (now & was) # ... the shared node stands + + # the other fiber is BYTE-IDENTICAL — same rows, same currency + assert bench.memberships.fiber("x.py", "x1") == x_before + assert all(m.current for m in x_before) + assert bench.memberships.n_doc(shared_id) == 2 # ...and x.py still holds the shared atom + + # both lineages traverse the shared node; only y's slot minted an edge + assert bench.memberships.slot_runs("x.py", ["x1"])["shared"] == [shared_id] + assert bench.memberships.slot_runs("y.py", ["y1", "y2"])["shared"] == [shared_id] + assert bench.memberships.slot_edges("y.py", ["y1", "y2"])["only_y"] == [ + (atom_id(only_y), atom_id(edited))] + assert bench.memberships.slot_edges("x.py", ["x1"]) == {"shared": [], "only_x": []} + assert atom_id(only_x) in bench.memberships.atom_ids_of_path("x.py") + + +# ── Item 5 / §8(d) — the read path: default is current-view, history is opt-in ──────────── + +def test_retrieval_resolves_current_occupancies_and_history_is_opt_in(bench: Bench) -> None: + """§8(d). *Degenerate input:* an all-current fixture passes any current-filter vacuously, so a + superseded occupancy is asserted to EXIST before anything is read.""" + shared, _only_x, only_y = _fork_bench(bench) + edited = _chunk("only_y", "def only_y():\n return 99\n", span=(4, 5)) + bench.lander.land("y.py", "y2", [shared, edited]) + + # PRECONDITION: a superseded occupancy exists, and an atom whose EVERY home is superseded + superseded = [m for m in bench.memberships.occupancies(atom_id(only_y), + include_superseded=True) + if not m.current] + assert superseded, "fixture holds no superseded occupancy — the filter test would be vacuous" + assert bench.memberships.n_doc(atom_id(only_y)) == 0 + assert current_any_drift(bench.vectors, bench.memberships) == [] + + # the atom whose only home is superseded has current_any=false and drops out of the default + # search; `include_superseded=True` lifts the prefilter and surfaces it. + q = bench.embedder.embed_query(only_y.text) + default_hits = bench.vectors.search(q, k=50, provenances={Provenance.CODE}) + assert atom_id(only_y) not in {str(h["id"]) for h in default_hits} + with_history = bench.vectors.search(q, k=50, provenances={Provenance.CODE}, + include_superseded=True) + assert atom_id(only_y) in {str(h["id"]) for h in with_history} + + # the JOIN: default consumers see current occupancies only... + resolved = resolve_occupancies(bench.memberships, default_hits) + assert resolved and all(o.current for r in resolved for o in r.occupancies) + # ...and the opt-in surfaces the superseded one + deep = resolve_occupancies(bench.memberships, with_history, include_superseded=True) + assert any(not o.current for r in deep for o in r.occupancies) + + # a SHARED atom resolves to all its current homes — one hit, several places (the D3 feature) + shared_hit = [h for h in with_history if str(h["id"]) == atom_id(shared)] + assert shared_hit + (one,) = resolve_occupancies(bench.memberships, shared_hit) + assert {o.path for o in one.occupancies} == {"x.py", "y.py"} + assert len(one.occupancies) == 2 + + +# ── Item 6 / §8(c) — append-only, and the one recorded hole ────────────────────────────── + +def test_purge_is_the_only_vector_removal_and_it_leaves_a_recorded_hole(bench: Bench) -> None: + """§8(c). *Degenerate input:* a purge that silently no-ops also satisfies "never deletes" — so + the assertions are that the hole EXISTS: the row is gone, the memberships are tombstoned (not + deleted), and every note-lane removal path is shown to act on a note row while leaving |V| + untouched.""" + shared, _only_x, _only_y = _fork_bench(bench) + bench.vectors.add([{"id": "n:0", "digest": "noteD", "title": "N", + "source_path": "notes/n.md", "chunk_index": 0, + "provenance": Provenance.AUTHORED_SOLO.value, "text": "a note", + "vector": [0.3] * DIM}]) + v_before = bench.n_vectors() + m_before = bench.memberships.count() + assert v_before == 3 and m_before == 4 # PRECONDITION: something to lose + + # the note-lane removal paths ACT (so this is not "the calls were no-ops")... + bench.vectors.delete_source("notes/n.md") + assert len(bench.vectors.all_rows(provenances={Provenance.AUTHORED_SOLO})) == 0 + bench.vectors.delete(digest="noteD") + bench.vectors.delete_source("x.py") # a path an atom row cannot carry + # ...and reach NO atom row: `source_path`/`digest` are shed, so the predicates cannot match + assert bench.n_vectors() == v_before + assert bench.memberships.count() == m_before + + report = purge_atom(bench.vectors, bench.memberships, atom_id(shared)) + + assert report.vector_rows_deleted == 1 # ← the purge OBSERVABLY acted + assert report.memberships_tombstoned == 2 # ...on both of the shared atom's homes + assert bench.n_vectors() == v_before - 1 # the ONE admitted |V| decrease + assert atom_id(shared) not in {str(r["id"]) for r in bench.vectors.atom_rows()} + + # the hole is RECORDED, not silent: the occupancies survive, tombstoned and not current + holes = bench.memberships.occupancies(atom_id(shared), include_superseded=True) + assert len(holes) == 2 and all(h.tombstoned and not h.current for h in holes) + assert bench.memberships.count() == m_before # nothing was deleted from M + assert bench.memberships.n_doc(atom_id(shared), current_only=False) == 0 + + +# ── Item 6 / §8(e) — the D8 crash property ─────────────────────────────────────────────── + +def test_a_crash_between_the_vector_insert_and_the_fiber_write_repairs_on_reland( + bench: Bench, monkeypatch: pytest.MonkeyPatch) -> None: + """§8(e). *Degenerate input:* a "crash" injected AFTER all writes makes repair vacuous — so the + injection point is asserted (|V| grew, |M| did not), and the orphan is asserted OBSERVABLE + before anything is repaired. "Nothing dangles" is also true of a store that never wrote.""" + path = "pkg/w.py" + chunks = derive_code_chunks(path, _SRC_A) + v_before, m_before = bench.n_vectors(), bench.memberships.count() + + def boom(_rows: object) -> int: + raise RuntimeError("crash: power lost between the vector insert and the fiber write") + + monkeypatch.setattr(bench.memberships, "write_fiber", boom) + with pytest.raises(RuntimeError): + bench.lander.land(path, "blobA", chunks) + monkeypatch.undo() + + # THE INJECTION POINT, asserted: vectors landed, the fiber did not + assert bench.n_vectors() > v_before + assert bench.memberships.count() == m_before + assert bench.memberships.fiber(path, "blobA") == [] + + # THE ORPHAN IS OBSERVABLE — dormant geometry with no occupancy, and invisible to the default + # search because `current_any` is false at insert (D8's write order is what makes this safe) + orphans = bench.memberships.orphan_atom_ids() + assert orphans and orphans == {atom_id(c) for c in chunks} + assert all(not bool(r["current"]) for r in bench.vectors.atom_rows()) + assert current_any_drift(bench.vectors, bench.memberships) == [] + + # the idempotent re-land REPAIRS: it adopts the orphans (zero re-embeds) and nothing dangles + embeds_before = len(bench.embedder.embedded) + report = bench.lander.land(path, "blobA", chunks) + assert report.atoms_embedded == 0 # adopted, not re-embedded + assert len(bench.embedder.embedded) == embeds_before + assert bench.memberships.orphan_atom_ids() == set() + assert len(bench.memberships.fiber(path, "blobA")) == len(chunks) + assert all(m.current for m in bench.memberships.fiber(path, "blobA")) + assert current_any_drift(bench.vectors, bench.memberships) == [] + + +def test_repair_current_any_rebuilds_the_cache_from_membership_truth(bench: Bench) -> None: + """R3: `current_any` is a CACHE over membership truth, with a rebuild path. Seeded drift first, + so "the repair worked" is not a statement about a store that was never wrong.""" + shared, _x, _y = _fork_bench(bench) + assert current_any_drift(bench.vectors, bench.memberships) == [] # PRECONDITION: no drift yet + + bench.vectors.set_current_any([atom_id(shared)], False) # seed the drift by hand + assert current_any_drift(bench.vectors, bench.memberships) == [atom_id(shared)] + + raised, lowered = repair_current_any(bench.vectors, bench.memberships) + assert (raised, lowered) == (1, 0) + assert current_any_drift(bench.vectors, bench.memberships) == [] + + +# ── Item 7 / §8(f) — the §4 invariants on the degenerate fixture ───────────────────────── + +@dataclass +class Fixture: + bench: Bench + chain: list[str] + side_branch: str + shared_id: str + dup_id: str + v_trace: list[int] + + +@pytest.fixture +def degenerate(bench: Bench) -> Fixture: + """The §8(f) fixture, carrying ALL FOUR shapes — a fixture missing any one makes the matching + invariant vacuous: a REVERT (so adjacent-collapse is distinguishable from distinct-collapse), a + MERGE SIDE-BRANCH version (so "chain members ⊊ ledger versions" bites), a SHARED atom (so + `n_doc` is not always 1), and a DUPLICATE L0b WINDOW PAIR (so the multiset reading bites).""" + path = "pkg/w.py" + v_trace: list[int] = [] + + shared = _chunk("shared", "def shared():\n return 7\n", span=(1, 2)) + work_a = _chunk("work", "def work():\n return 1\n", span=(4, 5)) + work_b = _chunk("work", "def work():\n return 2\n", span=(4, 5)) + work_s = _chunk("work", "def work():\n return 3 # side branch\n", span=(4, 5)) + dup = _chunk("", "x = 1\n", layer=LAYER_CODE_TEXT, span=(7, 7)) + + for blob, chunks in (("A", [shared, work_a, dup, dup]), + ("B", [shared, work_b, dup, dup]), + ("S", [shared, work_s, dup, dup]), # the side branch (off-chain) + ("A", [shared, work_a, dup, dup])): # ← the revert + bench.lander.land(path, blob, chunks, head_blob_sha=blob) + v_trace.append(bench.n_vectors()) + # the shared atom also lives in a SECOND path, so n_doc > 1 somewhere + bench.lander.land("other.py", "o1", [shared]) + v_trace.append(bench.n_vectors()) + # HEAD is A (the revert), reconciled last so the store is in its converged state + bench.lander.reconcile(path, "A") + return Fixture(bench=bench, chain=["A", "B", "A"], side_branch="S", + shared_id=atom_id(shared), dup_id=atom_id(dup), v_trace=v_trace) + + +def test_the_fixture_carries_all_four_shapes(degenerate: Fixture) -> None: + """The precondition for everything below. A fixture missing a shape makes its invariant + vacuous, so the shapes are asserted once, here, rather than assumed four times.""" + m = degenerate.bench.memberships + path = "pkg/w.py" + work = m.slot_runs(path, degenerate.chain)["work"] + assert len(work) == 3 and work[0] == work[2] != work[1] # a REVERT exists + assert (path, "S") in m.fibers() and "S" not in degenerate.chain # a side branch exists + assert m.n_doc(degenerate.shared_id) >= 2 # a shared atom exists + assert len([x for x in m.fiber(path, "A") + if x.content_id == degenerate.dup_id]) == 2 # a duplicate pair + + +def test_per_slot_edges_equal_runs_minus_one_and_a_revert_is_not_collapsed( + degenerate: Fixture) -> None: + """§4's first invariant, and the C1 formulation that gives it content. + + `|edges| = |runs| - 1` is true of any consecutive-pair construction — what is FALSIFIABLE is + the collapse rule: A → B → A must give 3 runs and 2 edges (adjacent collapse), not 2 and 1 + (distinct collapse). The seeded violation is distinct-collapse over the same chain.""" + m = degenerate.bench.memberships + path, chain = "pkg/w.py", degenerate.chain + runs, edges = m.slot_runs(path, chain), m.slot_edges(path, chain) + + assert set(runs) == {"shared", "work"} # L0b is chainless (R4), so `dup` is out + for slot in runs: + assert len(edges[slot]) == len(runs[slot]) - 1 # ← the invariant + + # the revert survives: 3 runs, 2 edges, and the second edge RETURNS to the first occupant + assert len(runs["work"]) == 3 and len(edges["work"]) == 2 + assert edges["work"][0][1] == edges["work"][1][0] + assert edges["work"][1][1] == runs["work"][0] # ...back to A's occupant + # ...and it is a real distinction: distinct-collapse over the SAME chain gives 2 runs / 1 edge + assert len(dict.fromkeys(runs["work"])) == 2 + + # an unchanged slot chains as ONE run and NO edge — the "never mint a spurious edge" half + assert len(runs["shared"]) == 1 and edges["shared"] == [] + + +def test_a_side_branch_fiber_is_a_member_of_m_but_sits_on_no_chain(degenerate: Fixture) -> None: + """D4/F3: chains are first-parent, the ledger walks all commits — so a side-branch version is a + real member of M (it contributes to |M| and to n(v)) and contributes to NO edge count. + Quantifying the edge invariant over all fibers instead of chain members is the error.""" + m = degenerate.bench.memberships + path = "pkg/w.py" + side = m.fiber(path, degenerate.side_branch) + side_only = {x.content_id for x in side} - { + x.content_id for b in degenerate.chain for x in m.fiber(path, b)} + assert side_only, "PRECONDITION: the side branch holds an atom no chain fiber holds" + + assert sum(len(m.fiber(path, b)) for b in {*degenerate.chain, degenerate.side_branch}) \ + <= m.count() + for cid in side_only: + assert m.n_occ(cid, current_only=False) > 0 # it IS in M + assert all(cid not in run + for run in m.slot_runs(path, degenerate.chain).values()) # ...on no chain + + +def test_fiber_sizes_sum_to_the_membership_count(degenerate: Fixture) -> None: + """§4: Σ fiber sizes = |M| — the fibration is a PARTITION, so no occupancy is outside a fiber + and none is double-counted. The seeded violation is a `fiber()` that quietly drops superseded + rows, which is the realistic way this breaks: the fixture holds superseded fibers (asserted), + so the mutation genuinely bites.""" + m = degenerate.bench.memberships + assert sum(len(m.fiber(p, b)) for p, b in m.fibers()) == m.count() + + superseded = [(p, b) for p, b in m.fibers() if not any(x.current for x in m.fiber(p, b))] + assert superseded, "PRECONDITION: without a superseded fiber the mutation below is a no-op" + + class _CurrentOnlyFiber(MembershipStore): + def fiber(self, path: str, blob_sha: str) -> list[Membership]: + return [x for x in super().fiber(path, blob_sha) if x.current] + + mutated = _CurrentOnlyFiber(m.path) + assert sum(len(mutated.fiber(p, b)) for p, b in mutated.fibers()) < mutated.count() + mutated.close() + + +def test_current_any_is_equivalent_to_a_positive_current_document_count( + degenerate: Fixture) -> None: + """§4/R3: `current_any(v) ⇔ n_doc(v, t) > 0`. The fixture holds BOTH kinds of atom (asserted), + so the equivalence is not satisfied by a store where every atom is current.""" + b = degenerate.bench + rows = b.vectors.atom_rows() + assert any(bool(r["current"]) for r in rows) and any(not bool(r["current"]) for r in rows) + + for r in rows: + assert bool(r["current"]) == (b.memberships.n_doc(str(r["id"])) > 0) + assert current_any_drift(b.vectors, b.memberships) == [] + + # seeded violation: one flag flipped by hand and the ratchet names exactly that atom + victim = str(rows[0]["id"]) + b.vectors.set_current_any([victim], not bool(rows[0]["current"])) + assert current_any_drift(b.vectors, b.memberships) == [victim] + + +def test_the_vector_plane_never_shrinks_except_across_a_purge(degenerate: Fixture) -> None: + """§4: append-only ⇒ no test may ever observe |V| decrease, except across a LOGGED purge. The + trace is recorded while the fixture is built, so this reads real landings (including a revert + and a side branch) rather than a store nothing ever wrote to.""" + b = degenerate.bench + trace = degenerate.v_trace + assert len(trace) >= 4 and trace[0] > 0 # PRECONDITION: real landings happened + assert trace == sorted(trace) # monotone ↑ across every land + assert trace[-1] > trace[0] # ...and it genuinely grew + + before = b.n_vectors() + b.lander.reconcile("pkg/w.py", "A") # currency work never removes geometry + assert b.n_vectors() == before + + report = purge_atom(b.vectors, b.memberships, degenerate.shared_id) + assert report.vector_rows_deleted == 1 # the ONE exception, and it is logged + assert b.n_vectors() == before - 1