diff --git a/core/stores/memberships.py b/core/stores/memberships.py index 38018fe..b76ddcf 100644 --- a/core/stores/memberships.py +++ b/core/stores/memberships.py @@ -49,7 +49,7 @@ import sqlite3 from collections.abc import Iterable, Sequence -from dataclasses import dataclass +from dataclasses import dataclass, field from datetime import UTC, datetime from pathlib import Path @@ -343,6 +343,65 @@ def n_occ(self, content_id: str, *, current_only: bool = True) -> int: f"WHERE content_id = ? AND tombstoned = 0 {clause}", [content_id]).fetchone() return int(row[0]) if row else 0 + def n_doc_counts(self, *, layer: str | None = None, + current_only: bool = False) -> dict[str, int]: + """`n_doc(v)` for EVERY atom at once — one GROUP BY instead of |V| point queries. + + Defaults to the LIFETIME reading (`current_only=False`), because that is the variant the + rank-frequency histogram is defined over (D6): the corpus's vocabulary shape is a property + of everything it has ever held, not of one cut. Atoms with no occupancy do not appear — + their `n_doc` is 0 by definition, and materializing a zero per orphan would make the + histogram's tail an artifact of the ledger rather than of the corpus.""" + clause = "AND current = 1" if current_only else "" + lane = "AND layer = ?" if layer else "" + params = [layer] if layer else [] + return {str(r["content_id"]): int(r["n"]) for r in self._conn.execute( + f"SELECT content_id, count(DISTINCT path) AS n FROM memberships " + f"WHERE tombstoned = 0 {clause} {lane} GROUP BY content_id", params).fetchall()} + + def rank_frequency(self, *, layer: str | None = None, + current_only: bool = False) -> list[int]: + """The rank-frequency histogram of lifetime `n_doc(v)`: frequencies sorted DESCENDING, so + index *i* is rank *i+1* (D6, §4). + + Returned as plain data rather than a plot: Zipf conformance is a falsifiable corpus + property and must be CHECKED, never assumed (T2/T4), and a caller that wants the check + needs the numbers. The gauge earns its keep only if a shape anomaly localizes something + real — boilerplate consolidation, vocabulary flux; if it never does, D6 says cut it.""" + return sorted(self.n_doc_counts(layer=layer, current_only=current_only).values(), + reverse=True) + + def occupied_atoms(self, *, layer: str | None = None) -> int: + """Distinct atoms with at least one (un-tombstoned) occupancy — the denominator that makes + `embeds_avoided` honest. NOT the same as `|V|`: the plane may hold an orphan atom that no + fiber references (D8's crash window), and counting it as occupied would understate the + reuse this store is here to measure.""" + lane = "AND layer = ?" if layer else "" + params = [layer] if layer else [] + row = self._conn.execute( + f"SELECT count(DISTINCT content_id) FROM memberships WHERE tombstoned = 0 {lane}", + params).fetchone() + return int(row[0]) if row else 0 + + def occupancy_count(self, *, layer: str | None = None) -> int: + """`|M|` restricted to one lane (or the whole relation) — the numerator of `|M|/|V|`.""" + lane = "WHERE layer = ?" if layer else "" + params = [layer] if layer else [] + row = self._conn.execute( + f"SELECT count(*) FROM memberships {lane}", params).fetchone() + return int(row[0]) if row else 0 + + def lane_gauges(self) -> dict[str, LaneGauge]: + """Per-layer `(|M|, atoms)` in ONE aggregate query — the per-lane half of `|M|/|V|` (D6). + + Kept as a store method rather than a query written at the gauge site so the two counts are + taken at the same cut, from the same scan: computing them separately is how a `|M|` from + after a landing gets divided by a `|V|` from before it.""" + return {str(r["layer"]): LaneGauge(occupancies=int(r["n"]), atoms=int(r["v"])) + for r in self._conn.execute( + "SELECT layer, count(*) AS n, count(DISTINCT content_id) AS v " + "FROM memberships WHERE tombstoned = 0 GROUP BY layer").fetchall()} + def atom_ids_of_path(self, path: str) -> set[str]: return {str(r["content_id"]) for r in self._conn.execute( "SELECT DISTINCT content_id FROM memberships WHERE path = ?", [path]).fetchall()} @@ -484,6 +543,80 @@ def resolve_occupancies(memberships: MembershipStore, hits: Sequence[dict[str, o return out +@dataclass(frozen=True) +class LaneGauge: + """One lane's frequency-plane reading.""" + + occupancies: int = 0 # |M| restricted to this lane + atoms: int = 0 # distinct atoms holding an occupancy here + + @property + def dedup_factor(self) -> float: + """Occupancies per atom — how much reuse this lane is actually buying.""" + return (self.occupancies / self.atoms) if self.atoms else 0.0 + + @property + def embeds_avoided(self) -> int: + """Occupancies past the first for each atom: the embeds the duplicated model would have + paid and this one does not.""" + return max(0, self.occupancies - self.atoms) + + +@dataclass(frozen=True) +class FrequencyGauges: + """The D6 standing gauges: `|M|`, `|V|`, the dedup factor, and embeds-avoided — per lane and + over the whole plane. + + ⚑ **`dedup_factor` IS the D7 falsifier, kept observable forever rather than measured once** + (the S5 amendment). It fails its keep by sitting at ≈1.0 after a full rebuild: that would mean + the membership model bought nothing and D7's economics are false. Reading ≈1.0 is therefore not + "a low number", it is the design being wrong, and the gauge exists to say so out loud. + + `plane_atoms` is `|V|` — every atom row in the plane — while `atoms` counts only the atoms some + fiber references. They differ by exactly the orphans (D8's crash window), and keeping them + separate is what stops a repair-pass bug from quietly moving the dedup factor.""" + + occupancies: int = 0 + atoms: int = 0 + plane_atoms: int = 0 + per_layer: dict[str, LaneGauge] = field(default_factory=dict) + + @property + def dedup_factor(self) -> float: + """`|M|/|V|` (D6) — occupancies per atom in the plane.""" + return (self.occupancies / self.plane_atoms) if self.plane_atoms else 0.0 + + @property + def embeds_avoided(self) -> int: + return max(0, self.occupancies - self.atoms) + + @property + def orphans(self) -> int: + return max(0, self.plane_atoms - self.atoms) + + def __str__(self) -> str: + lanes = " · ".join(f"{k} {v.dedup_factor:.2f}×" for k, v in sorted(self.per_layer.items())) + return (f"|M|={self.occupancies} |V|={self.plane_atoms} " + f"dedup={self.dedup_factor:.2f}× ({lanes}) " + f"embeds_avoided={self.embeds_avoided} orphans={self.orphans}") + + +def frequency_gauges(vectors: VectorStore, memberships: MembershipStore) -> FrequencyGauges: + """Read the D6 gauges. Cheap and read-only: three aggregate SQL queries plus one server-side + row count — no vector crosses into Python, which is what makes this registrable on a cadence + beside the drift-gauge family rather than an occasional investigation. + + Every figure here is a QUERY. That is the issue #28 defect class stated as a rule: this repo + already carries docstrings quoting an edge count 8.4× off the live store, and a gauge whose + numbers were baked in at authoring time is that same defect with a dashboard on it.""" + return FrequencyGauges( + occupancies=memberships.count(), + atoms=memberships.occupied_atoms(), + plane_atoms=vectors.atom_row_count(), + per_layer=memberships.lane_gauges(), + ) + + def repair_current_any(vectors: VectorStore, memberships: MembershipStore) -> tuple[int, int]: """Rebuild the `current_any` cache from membership truth (D8/R3). Returns (raised, lowered). diff --git a/core/stores/vectorstore.py b/core/stores/vectorstore.py index 6047d84..7dcc299 100644 --- a/core/stores/vectorstore.py +++ b/core/stores/vectorstore.py @@ -10,8 +10,9 @@ from __future__ import annotations -from collections.abc import Iterable +from collections.abc import Iterable, Sequence from dataclasses import dataclass +from datetime import timedelta from pathlib import Path from typing import Any @@ -90,6 +91,36 @@ def _schema(dim: int) -> pa.Schema: } +@dataclass(frozen=True) +class CompactionReport: + """What `VectorStore.compact` physically did (§3). + + Both a BEFORE and an AFTER row count are carried on purpose. Compaction's whole contract is + that it is semantically invisible, and the only way to assert invisibility without asserting it + vacuously is to have the two numbers side by side and require them equal WHILE the version + count fell — a compaction that no-ops leaves every number equal, including the one that is + supposed to move.""" + + rows_before: int = 0 + rows_after: int = 0 + versions_before: int = 0 + versions_after: int = 0 + + @property + def rows_preserved(self) -> bool: + """The invariant: compaction removes no logical row.""" + return self.rows_before == self.rows_after + + @property + def versions_dropped(self) -> int: + return max(0, self.versions_before - self.versions_after) + + def __str__(self) -> str: + return (f"rows {self.rows_before}->{self.rows_after} " + f"versions {self.versions_before}->{self.versions_after} " + f"({self.versions_dropped} dropped)") + + def is_code_atom_row(row: dict[str, Any]) -> bool: """Is this row a shed CODE **atom** row (D1) rather than a source-object chunk row? @@ -236,6 +267,51 @@ def rows_for_source(self, source_path: str) -> list[dict[str, Any]]: self._table().scan().where(f"source_path = {_sql_str(source_path)}") .limit(0).to_list()] + def project(self, columns: Sequence[str], *, where: str | None = None, + limit: int = 0) -> list[dict[str, Any]]: + """A projected, predicate-pushed read — the named columns and nothing else. + + The generalization of `rows_for_source`' shape, and it exists for one measurable reason: + `vector` is 2560 floats per row, so a scan that does not name it costs a small fraction of + one that does. The rebuild's baseline (`ops/code_rebuild.py`) reads `id`/`layer`/`text` + over the whole code lane to re-derive the dedup economics and must never pay for geometry + it does not read; the carry-forward seed names `vector` precisely because copying it is the + point. `limit(0)` means UNLIMITED (verified empirically against the installed 0.33.0 and + pinned by the shim ratchet, exactly as `rows_for_source` documents). + + `where` is a raw LanceDB predicate and is the CALLER's to build — use `_sql_str` for any + value that is not a literal this module wrote itself.""" + if TABLE not in self._db.list_tables().tables: + return [] + q = self._table().scan().select(list(columns)) + if where is not None: + q = q.where(where) + return [dict(r) for r in q.limit(limit).to_list()] + + def supersede_legacy_code_rows(self) -> int: + """Flip every PRE-D1 duplicated code row to `current=false`, retaining it. Returns rows + flipped (0 when the store holds none — so a second call is a no-op). + + The predicate is the exact complement of `is_code_atom_row` within the code lane: a row is + legacy iff it is CODE and still carries the occupancy coordinates an atom row sheds. The + rebuild lands the atom plane into the same table as the rows it replaces, so without this + both models answer the default current-view search and the dedup is bought but not served. + + This is keep-and-link (`supersede_source`'s mechanism, D2) pointed at the retired ROW MODEL + rather than at a superseded version: one pushed-down predicate, a filtered count, one + in-place `update`. **Nothing is deleted** — `|V|` cannot decrease here, and D5's "purge is + the ONE removal" is untouched — and because the rows remain, the step is reversible by the + same update in the other direction.""" + if TABLE not in self._db.list_tables().tables: + return 0 + table = self._table() + where = (f"provenance = {_sql_str(Provenance.CODE.value)} " + "AND source_path <> '' AND current = true") + flipped = table.count_rows(where) # portable: do NOT rely on UpdateResult (bp-103 §11) + if flipped: + table.update(where, {"current": False}) + return flipped + def delete_source(self, source_path: str) -> None: """Drop every derived row for one source document, by `source_path` (the stable doc identity an amendment replaces a projection under — §4). Idempotent. @@ -367,6 +443,20 @@ def atom_rows(self) -> list[dict[str, Any]]: logged purge).""" return [r for r in self.all_rows(provenances={Provenance.CODE}) if is_code_atom_row(r)] + def atom_row_count(self) -> int: + """`|V|` — how many shed CODE-atom rows the plane holds, as a SERVER-SIDE count. + + The same number `len(atom_rows())` gives, without the scan: `atom_rows` materializes every + row INCLUDING its 2560-float vector, which is the right cost for the repair pass (it reads + the flag on each row) and entirely the wrong cost for a gauge that runs on a cadence. The + predicate is `is_code_atom_row` written as SQL — the one place those two spellings must + agree, which is why the Python predicate is the documented reading and this cites it.""" + if TABLE not in self._db.list_tables().tables: + return 0 + return self._table().count_rows( + f"provenance = {_sql_str(Provenance.CODE.value)} " + "AND source_path = '' AND digest = ''") + def set_current_any(self, ids: Iterable[str], value: bool) -> int: """Set `current_any` on exactly the named atom rows (D2 step 5). Returns rows written. @@ -407,6 +497,46 @@ def delete_atom(self, content_id: str) -> int: table.delete(where) return n + def dataset_versions(self) -> int: + """How many dataset versions the table still retains (§3). One per write batch, and D2 makes + `current_any` flips routine — so this is the number compaction is measured against.""" + if TABLE not in self._db.list_tables().tables: + return 0 + return len(self._table().list_versions()) + + def compact(self, *, older_than: timedelta | None = None) -> CompactionReport: + """Physical maintenance: compact fragments, then drop old dataset versions (§3). + + [cross-ref: extension] No compaction path existed in this module (the note verified it), and + §3 makes it part of the store's semantics rather than an operational afterthought: the lance + dataset accumulates a version per write batch — 298 versions / 245 MB measured 2026-07-27 + against ~232 MB of raw payload — and `current_any` flips are updates that rewrite fragments. + The rebuild ends with this, and housekeeping runs the cleanup half on cadence. + + **Compaction is PHYSICAL, never logical.** It removes no row and changes no vector, so row + count and search results are invariant across it — the acceptance asserts BOTH, because + "nothing changed" is also what a compaction that silently did nothing produces. The version + count dropping is the other half, and it is the half that makes the assertion non-vacuous. + + `older_than=None` takes the package's own retention default. Passing `timedelta(0)` reclaims + every superseded version immediately, which is what the rebuild wants (it has just written + thousands of batches) and what a test needs to observe a drop at all — but it forfeits + time-travel to any earlier version, so it is the caller's explicit choice, never the + default. `current_any` stays in lance throughout: the ANN prefilter needs it (D1/§3).""" + if TABLE not in self._db.list_tables().tables: + return CompactionReport() + table = self._table() + before_rows, before_versions = table.count_rows(None), len(table.list_versions()) + # ONE call does both halves. The deprecated `compact_files`/`cleanup_old_versions` pair + # routes through `Table.to_lance()` and raises ImportError without the optional `pylance` + # package (found by running it, not by reading about it) — see the shim's note. + table.optimize(cleanup_older_than=older_than, delete_unverified=False) + after_rows, after_versions = table.count_rows(None), len(table.list_versions()) + return CompactionReport( + rows_before=before_rows, rows_after=after_rows, + versions_before=before_versions, versions_after=after_versions, + ) + def search(self, vector: list[float], *, k: int = 5, provenances: Iterable[Provenance] | None = None, include_superseded: bool = False) -> list[dict[str, Any]]: diff --git a/core/typedshims/lancedb.py b/core/typedshims/lancedb.py index b323a45..42e9b71 100644 --- a/core/typedshims/lancedb.py +++ b/core/typedshims/lancedb.py @@ -29,6 +29,7 @@ from __future__ import annotations from collections.abc import Mapping, Sequence +from datetime import timedelta from typing import Any, Protocol import lancedb # type: ignore[import-untyped] # warrant: no py.typed upstream (V2); Any quarantined to this shim @@ -96,6 +97,30 @@ def search(self, vector: list[float]) -> VectorQuery: ... def scan(self) -> VectorQuery: ... + # [cross-ref: extension] bp-153 brings the PHYSICAL maintenance calls + # (dn-vector-membership-store §3): the dataset accumulates a version per write batch, and D2 + # makes `current_any` flips routine, so the rebuild ends with compaction + old-version cleanup + # and housekeeping runs cleanup on cadence. + # + # ⚑ The note's §3 Q5 splits into two cases and only one is a finding. Widening THIS Protocol is + # ordinary in-scope work — the shim is our code. The finding case would be the pinned lancedb + # release having no such capability underneath, where widening conjures nothing. It does NOT + # apply, and the check is the bp-103 rule (read the INSTALLED package, never the docs): 0.33.0's + # `lancedb.table.Table` exposes `optimize` and `list_versions`, both verified to run here. + # + # `optimize` is named rather than the `compact_files` + `cleanup_old_versions` pair the API also + # carries, for a reason found by running it: BOTH of those are deprecated as of 0.21.0 and route + # through `Table.to_lance()`, which raises `ImportError` unless the optional `pylance` package + # is installed — not a dependency of this project, and adding one to reach a deprecated path + # would be the wrong direction. `optimize` (modeled on PostgreSQL's VACUUM) does both halves in + # one call, on the supported path, with no extra package. Measured on a 7-version table: 7 -> 1 + # versions, row count unchanged. + + def optimize(self, *, cleanup_older_than: timedelta | None = ..., + delete_unverified: bool = ..., retrain: bool = ...) -> None: ... + + def list_versions(self) -> list[dict[str, object]]: ... + class VectorDB(Protocol): """The slice of a LanceDB connection the store calls.""" @@ -155,6 +180,24 @@ def scan(self) -> VectorQuery: q: VectorQuery = self._raw.search(None) return q + def optimize(self, *, cleanup_older_than: timedelta | None = None, + delete_unverified: bool = False, retrain: bool = False) -> None: + """Compact fragments and drop superseded dataset versions — the VACUUM-shaped call (§3). + + PHYSICAL only: it removes no logical row, so row count and search results are invariant + across it — the property the compaction test asserts rather than assumes. A `None` window + takes the package's own retention default; choosing a window is the caller's policy + (`VectorStore.compact`), never the boundary shim's. Returns nothing in 0.33.0, which is why + the store measures the effect with `list_versions` instead of trusting a return value.""" + self._raw.optimize(cleanup_older_than=cleanup_older_than, + delete_unverified=delete_unverified, retrain=retrain) + + def list_versions(self) -> list[dict[str, object]]: + """Every dataset version the table still retains — the measurement that makes "the version + count DROPPED" checkable instead of asserted.""" + versions: list[dict[str, object]] = self._raw.list_versions() + return versions + class _DB: """Runtime adapter for a raw lancedb connection — structurally a `VectorDB`. diff --git a/docs/build-plans/bp-153/journal.md b/docs/build-plans/bp-153/journal.md index 8009ff6..96dff00 100644 --- a/docs/build-plans/bp-153/journal.md +++ b/docs/build-plans/bp-153/journal.md @@ -1,76 +1,349 @@ # bp-153 — journal -## Pre-build notes for whoever picks this up - -- ⚑⚑ **The old duplicated backfill must NEVER be run.** 52,755 embeds vs 22,502 atoms is - **2.34× measured waste** (D7). This is not "prefer the new path" — it is a prohibition. The - one historical attempt (job 300240) died in `TimeoutError` on 2026-07-25 and left - `_code_corpus_backfilled` at 0 rows. - -- ⚑⚑ **Clearing the wedge and draining the backlog is an OWNER op — you do not do it.** It - gates Item 3 only. If it is not done, **park Item 3 with its re-entry condition and - continue Items 1, 2** — never block the session on an owner action. (Standing rule: never - block on the owner; only a `blocker` finding ends a session early, and the Stop gate still - wants a fresh journal.) - -- ⚑⚑ **Never stop the daemon.** D7 is explicit: the rebuild runs as background queue jobs - with `jobs.checkpoint` resume tokens and a per-slice time budget, respecting single-writer - as a queue citizen. A rebuild that requires a daemon stop is a spec defect, not an - operational choice. Note the job class you are enlarging is exactly the one that wedged — - `code_sync` is the daemon's most recent failure (`TimeoutError`, tracked as issue #18). - -- ⚑ **The baseline is a MOVING TARGET — re-derive it, do not trust the constant.** The note's - 22,502 / 52,755 figures were measured 2026-07-27 over 1,653 ledger versions. Live - `palace status` on 2026-08-01 reports **33,861 vector rows** against the 22,621 the note - measured — the corpus has grown. **The portable claim is the ratio (~2.34×), not the - absolute count.** Item 1 exists precisely to re-derive both before anything is spent. - -- ⚑ **Do not hardcode a measured figure into a docstring.** Open issue #28 is this exact - defect elsewhere in the tree: `reference_edges` docstrings claim ~272k edges while the live - store holds 2,284,272 — an 8.4× stale inline constant that becomes a silent design input. - **A count is a query, not a comment.** Every figure this plan emits must be computed. - -- ⚑ **Step 0 has genuinely never run.** `capture_commit_diffs` (`ops/code_lineage.py:112-130`) - is shipped but the live snapshots db has zero `commit_diffs` / `_commit_diffs_captured` - tables. Nothing downstream of it has ever executed against real data — treat its first run - as unproven machinery, not as a routine invocation. It *is* idempotent per commit - (`:119-129`, `INSERT OR IGNORE` + marker), which is what makes slicing safe. - -- ⚑ **The compaction shim has two failure cases and only one is a finding (§3 Q5).** The - typed `VectorTable` Protocol (`core/typedshims/lancedb.py:82-97`) declares - `add/count_rows/delete/update/to_arrow/search/scan` and no compaction member. **Widening - our own Protocol is ordinary in-scope work** — the file is in `write_scope`. **Discovering - the pinned lancedb release has no such capability underneath is the finding** — and then - you stop, because widening a Protocol cannot conjure a capability, and improvising a - physical rewrite against an unpinned surface is how you lose a corpus. - -- ⚑ **The embedder pin decides whether the carry-forward seed is free.** 13,311 atoms - (7,791 L0a + 5,520 L0b) are already embedded with embed text unchanged — but reuse is only - valid if their embedder identity matches the live `EmbeddingConfig.model` + `dim` - (`core/kernel/config/loader.py:138-141`). If the embedder changed since they landed, **the - seed is not free and the rebuild's economics change** — raise before spending, because the - owner blessed a cost that would no longer be true. L1 re-embeds regardless (3,424 atoms): - its stored windows were cut over header-bearing prose and do not survive D0's pin. - -- ⚑ **"|atoms| ≤ Σ chunks" is NOT acceptance.** It holds vacuously at zero savings — the - false-success shape the note calls out by name. The **measured factor** is the claim under - test. Likewise `n_doc` vs `n_occ`: a fixture without a duplicate L0b window pair cannot - distinguish them, so the fixture precondition is asserted first. - -- ⚑ **Fibers for EVERY ledger version, not just chain members.** The ledger walks all commits - (`ops/code_snapshot.py:353-360`, `rev-list --reverse HEAD`) while chains are first-parent - (`ops/code_lineage.py:85-95`). A side-branch version lands a fiber but sits on **no chain** - (D4/F3). Quantifying edge invariants over all fibers instead of chain members is wrong. - -- ⚑ **Prose is not migrated (PD-2/R5).** Note rows stay on the old path. Do not "helpfully" - migrate them; the D1 stratum fence has not been redesigned. - -- ⚑ **Never run `deploy`.** Promoting a run onto HEAD is the owner's single in-loop gate. - Also note `down`/`up`/`restart`/`deploy` live on `scripts/palace.py`, not on the - `mind-palace` wrapper. - -- **On completion the track is "ready to deskcheck", not done.** File it into - `docs/DESKCHECK-QUEUE.md` and say so. Never self-declare a track done. - -- **Depends on bp-152** — there must be a lander to rebuild into and a membership relation - for the gauges to count. +## Session 1 — 2026-08-12 · builder, worktree `build/bp-153-rebuild-gauges-probe-compaction` + +**Base:** `f5306d4` ("docs(bp-152): journal the membership-store build"), verified before any +write. bp-151, bp-155, bp-152 all merged. + +**Status: all seven items BUILT. One item's live execution is parked on an owner op and that is +the plan's design, not an incomplete item** — Item 3's machinery is complete and proven on +fixtures; its live run is `palace code-rebuild` after the deploy (re-entry condition below). +Three issues filed (#50, #51, #52); this plan closes #39. + +--- + +### The wedge state observed at build time + +`palace queue`, read-only, 2026-08-12: **0 queued · 0 running · 5 deferred** (3 `curate`, 2 +`dream`, deferred 15–18 days). Lifetime 534,735 done / 23 failed. **The wedge is drained** — the +D7/S6 owner precondition Item 3 was gated on is satisfied. The live daemon (run #40) deliberately +runs pre-bp-152 code, which is why the live rebuild physically cannot execute until the owner +deploys; issue #39 holds that deploy until this lands. + +Corroborating live state, all read-only: `data/memberships.sqlite` **does not exist** (bp-152 is +built but not deployed), the vector store holds **33,861** rows of which **33,823** are CODE and +**zero** are shed atom rows, and the snapshots ledger's newest commit is `bb0caa7` +(2026-07-28) — the ledger is frozen at that cut because the code lane is `enabled=False` on this +machine. + +--- + +### Item 1 — the read-only baseline, RUN FOR REAL against the live store + ledger + +The whole point of this item is that the design's economics were measured on a July cut and the +corpus grows, so **the ratio is the portable claim and the absolute counts are not**. + +| figure | design (2026-07-27, 1,653 versions) | **measured (2026-08-12, 1,663 versions)** | Δ | +|---|---|---|---| +| Σ per-version chunks (duplicated model) | 52,755 | **52,200** | −1.1% | +| distinct atoms under D0 | 22,502 | **22,897** | +1.8% | +| **ratio** | **2.34×** | **2.280×** | **−2.6%** | +| carry-forward seed | 13,311 (7,791 L0a + 5,520 L0b) | **15,186** (8,681 L0a + 6,505 L0b) | +14.1% | +| embeds avoided | 30,253 | **29,303** | — | + +**−2.6% is well inside the ~10% band, so the §8(g) falsifier did NOT fire and Item 3 proceeded on +the blessed economics.** Per lane: L0a **2.55×** (design 2.54×), L0b **2.06×** (2.05×), L1 +**1.96×** (2.42× — the one real deviation, filed as #51). The seed is **66.3%** of the atom set +(design projected ~59%). + +**Embedder identity — the pin, answered on every axis it has.** `stored_dim=2560` equals the live +`EmbeddingConfig.dim`, and `ledger_mismatched=0`. The `model` is **not recorded on a pre-D1 row** — +structurally: the Arrow schema is shared with the prose lane and has no embedder column, which is +precisely the gap bp-152's `atoms` ledger closes for everything landed hereafter. The report says so +(`unrecorded_rows=15186`) rather than assuming a match. The gap is closed by evidence outside the +store: `config/defaults.toml`'s `model = "qwen3-embedding:4b"` has **exactly one commit in its +entire history** (`7502109`, 2026-06-25), predating the code lane — so no model change can have +occurred under those rows. **Verdict: the seed's geometry is the live one; the carry-forward is +free.** + +**Read-only, verified rather than asserted:** the pass ran with a stub embedder that raises if +called; live store rows were **33,861 before and 33,861 after**; `|M|` stayed 0; no +`memberships.sqlite` was created beside the live vault catalog. Runtime 11.9s for 1,663 versions. + +⚑ **A trap for a successor:** `get_config()` resolves `data_dir` **relative to the CWD's repo +root**, so running this from a worktree measures the worktree's (empty) store, not the live one. +The first run did exactly that and reported `seed=0` from a 0-row store. Every live measurement +here uses explicit absolute paths. The same hazard is why the `code-rebuild` verb passes +`repo=self.repo_root` rather than taking `build_code_corpus_sync`'s CWD-derived default. + +Receipts: `ops/code_rebuild.py:measure_baseline`; `tests/unit/test_code_rebuild.py` +`test_the_baseline_measures_the_factor_and_writes_nothing`. + +--- + +### Item 2 — step 0: the first successful `commit_diffs` capture in the project's history + +Confirmed at HEAD before touching anything: the live snapshots db has **zero** +`commit_diffs` / `_commit_diffs_captured` tables. §3 Q1 and panel finding S2 hold — the machinery +has shipped since bp-099 and had never run. + +Proven on a **copy**, never the live db (the live daemon owns that file and single-writer is the +invariant). The copy carries the `snapshots` table verbatim in live rowid order — that is the +entire input `ledger_commits`/`supersession_chains` read, so the proof is against the REAL ledger's +commits without a 5 GB duplication or a write handle on the live file. + +- **1,377 commits captured in 4 budgeted slices, 18s**, at a deliberately tight 5s budget. +- **2,209 `commit_diffs` rows**, 1,377 marker rows. +- **Re-run captured 0** — idempotence, via the marker table. +- **The R6 falsifier did not fire.** Every slice that hit its budget yielded at a commit boundary + with its progress already durable. There is no window in which work is done but unrecorded, + because the durable progress IS `_commit_diffs_captured` — the budget is only ever checked + *between* commits, each of which is its own transaction. The 300240 `TimeoutError` failure mode + is bounded, not merely hoped away. + +**A real revert exists in this repo's own history**, which is the receipt the F4 dispute wanted: +`ops/lifecycle/launcher.py` re-occupies blob `e09f038e` at run positions **10 and 12** +(non-adjacent) and `281bce1d` at 9 and 11; `tests/unit/test_interpreter_versions.py` likewise. +Adjacent-collapse preserved every one of them; a distinct-collapse would have erased them. + +Receipts: `ops/code_rebuild.py:capture_slice`, `pending_commits`; +`tests/unit/test_code_rebuild.py::test_the_sliced_capture_is_idempotent_and_leaves_durable_progress`. + +--- + +### Item 3 — the sliced, checkpointed, resumable rebuild + +Built complete: `seed_carry_forward` (bulk embed-reuse by canonical re-hash, zero embedder calls), +`rebuild_slice` (the budgeted walk over every ledger version), `rebuild_step` (the four-phase +machine `capture → seed → land → compact`), and `retire_legacy_rows`. + +**Resumability is proven, and proven not to depend on the token.** The acceptance kills a run +mid-slice (a zero budget forces a yield with work already landed — asserted, so it is a real +resume), resumes from the token to completion, and compares against an **independent single-shot +rebuild of the same ledger**: identical `|V|`, identical `|M|`, identical fibers chunk-for-chunk. +Then the token is thrown away entirely and the walk re-run from scratch — **0 new occupancies, 0 +embeds** — and again with an unreadable token. Fiber equality holds because derivation is pure, so +the token buys time and nothing else. + +**§8(g) is asserted as an EQUALITY, not the inequality the design warns about.** `|atoms| ≤ Σ +chunks` holds vacuously at zero savings; instead the standing `|M|/|V|` gauge must reproduce the +baseline's ratio *exactly*, because `|M|` **is** Σ per-version chunks and `|V|` **is** the distinct +atom count — the same number reached from two independent directions, derived-and-counted before +the run and stored-and-queried after it. The fixture's sharing (across files AND across versions) is +asserted before any ratio is read. + +**`retire_legacy_rows` is a decision the plan did not enumerate**, reported separately for exactly +that reason. The rebuild lands the atom plane into the same table as the duplicated rows it +replaces, so both would answer the default current-view search — the dedup would be bought and then +not served. It is a **supersession** (keep-and-link, D2), never a removal: nothing deleted, `|V|` +cannot fall, D5's "purge is the ONE removal" untouched, and reversible by the same update in the +other direction. Flagged in the PR body for the merge audit. + +**LIVE RUN — parked, with its re-entry condition:** + +> **Wedge drained (observed depth 0 at build); the live rebuild runs post-deploy via +> `palace code-rebuild` — an OWNER op.** The live daemon runs pre-bp-152 code, so the rebuild +> physically cannot execute until the owner deploys (issue #39 holds the deploy until this plan +> merges). Order on the owner's side: merge → `deploy` → `palace code-rebuild --dry-run` to confirm +> the ratio at the then-current cut → `palace code-rebuild` to enqueue. The verb enqueues; the +> daemon drains it as checkpointed BACKGROUND slices. Expected at today's cut: ~22,897 atoms +> landed, ~15,186 of them free by carry-forward, so **~7,700 actual embeds** rather than the +> duplicated model's 52,200. + +--- + +### Item 4 — the frequency-plane gauges (D6) + +`n_doc_counts`, `rank_frequency`, `occupied_atoms`, `occupancy_count`, `lane_gauges` on the store; +`frequency_gauges(vectors, memberships) → FrequencyGauges` across the two. Three aggregate queries +plus one server-side row count — **no vector crosses into Python**, which is what makes it +registrable on a cadence rather than an occasional investigation. + +The F5 falsifier is exercised on the only input that can see it: `n_doc` and `n_occ` must **differ** +on a duplicate L0b window pair. The pair is asserted to exist as TWO rows with distinct +`chunk_index` inside ONE path *first* — then `n_occ == 2` against `n_doc == 1` (lifetime 6 vs 1) — +and a non-repeated atom is used as the control that makes the inequality mean something. The dedup +factor gets an explicit **control store** where every landing is a fresh atom and the gauge reads +≈1.0; without it "dedup > 1" would establish nothing. `dedup_factor` **is** the D7 falsifier kept +observable forever (the S5 amendment): it fails its keep by sitting at ≈1.0 after a rebuild. + +Receipts: `core/stores/memberships.py`; four new tests in `tests/unit/test_memberships.py`. + +--- + +### Item 5 — the probe re-home and the `:139` docstring correction — **closes #39** + +`_code_backfill_incomplete` (`ops/lifecycle/launcher.py:374-393`, call site `:551`) counted distinct +`(source_path, digest)` pairs over the code lane. bp-152 shed both columns from atom rows, so +against a rebuilt store every atom row collapses to the single tuple `('','')` and the probe reads +`1 < 1,663` on every daemon start — **enqueueing a backfill forever**. finding-0166's named +falsifier returning through a different door. + +The store side now reads `memberships.fibers()`: a version **is** its fiber, so this is the same +number at a sturdier home and the honest form of the F6 re-home. **Only the data source moved** — +cadence, call site, `ingestion.code.enabled` gate and enqueued kind all stand, pinned by its own +test, because a trigger-level change is out of design. + +The test builds a genuinely rebuilt store and **computes the old reading beside the new one**, +asserting the old one would have looped (`{('','')}`, `1 < 2`). Without that counterfactual the +assertion is only "the probe says complete", which the un-re-homed code could also produce on some +other input. It also asserts the probe still says INCOMPLETE when a version really is missing — +becoming a constant `False` is the other way to stop the loop, and it is useless. + +The `:139` docstring said a chain is "the ordered **distinct** sequence of its blobs" while the code +collapses only **adjacent** repeats — the exact distinction the F4 dispute rests on, so the prose +was contradicting the argument that cites it as evidence. Corrected, with a revert fixture that +computes both readings side by side and a ratchet asserting the contradicting phrase is **absent** +(gaining the word "adjacent" while keeping the wrong claim would otherwise still pass). + +--- + +### Item 6 — compaction and old-version cleanup (§3) + +**§3 Q5 resolves to its in-scope case, not the finding.** The capability exists underneath: verified +against the **installed** lancedb 0.33.0 (the bp-103 rule — read the package, not the docs). So +widening the `VectorTable` Protocol was ordinary work, and no finding was owed. + +**But the obvious path was wrong, and only running it showed that.** `compact_files` and +`cleanup_old_versions` — the pair the API advertises — are **deprecated as of 0.21.0 and route +through `Table.to_lance()`, which raises `ImportError` without the optional `pylance` package** +(not a dependency of this project). The shim declares **`optimize`** instead: one call, both halves, +supported path, no new dependency. Measured on a 7-version table: **7 → 1 versions, row count +unchanged**. + +Acceptance asserts both halves, since either alone is vacuous: row count AND search results +unchanged, **and** the version count actually **dropped** — with the fixture asserted to hold +several dataset versions first, or there is nothing to drop. The firewall gets its own integration +test: compaction and legacy-row retirement are physical rewrites and the firewall is a row +prefilter, so the notes are asserted retrievable before and after (identical hit ids), the code lane +still reachable through its own provenance set, and every row still carrying a non-empty +`provenance`. A firewall that held by deleting the corpus is not a firewall. + +--- + +### Item 7 — `palace code-rebuild`, the owner-visible verb + +Listed in `USAGE`, in the module docstring, and dispatched in `main()`; `CODE_REBUILD_KIND` + a +checkpointing handler registered unconditionally in `build_components`. **Wiring is the +deliverable** — a method nothing routes to is not a verb — so the tests read the dispatch out of the +*source* and assert the enqueued kind has a handler. + +This is the **first handler in the system to use the queue's checkpoint/resume protocol**, and the +one the protocol was built for: `checkpoint` clears the lease, so a yielded row reads as waiting +rather than orphaned and the next `claim` stamps a fresh **per-batch** deadline — the shape §2.10 +requires ("a healthy 14-hour backfill" must not die at hour N on a per-job deadline). + +The verb **enqueues**; it never stops the daemon, and a test reads the method's source for the calls +it must not make. `--dry-run` runs Item 1's read-only pass and writes nothing, asserted against a +real seeded ledger with an empty queue afterwards. + +--- + +### Gate (verbatim, run on the final tree) + +| leg | result | +|---|---| +| `uv run ruff check .` | **All checks passed!** (exit 0) | +| `uv run mypy core agents eval ops scheduler scripts` | **Success: no issues found in 265 source files** | +| `uv run mypy` (argless) | **Found 69 errors in 20 files (checked 570 source files)** — baseline **UNMOVED** | +| `uv run python -m ops.type_gate` | exit **0** (the one parked psutil report is pre-existing, non-fatal) | +| `uv run pytest -q` | **6 failed, 2515 passed, 15 skipped** in 429s | + +The 6 are the five known-red plus one named flake, and nothing else: + +1. `tests/e2e/test_dream_v2_live.py::test_dream_v2_synthesizes_grounded_themes_live` +2. `tests/integration/test_worktree_enforcement.py::test_a_deny_cross_worktree` +3. `tests/integration/test_worktree_enforcement.py::test_c_unsafe_direction_narrow_not_loosened` +4. `tests/integration/test_worktree_enforcement.py::test_d_no_pointer_is_no_plan_not_main_fallback` +5. `tests/unit/test_core_self_containment.py::test_core_imports_nothing_outside_core` +6. `tests/e2e/test_scheduler_live.py::test_supervisor_dispatches_a_real_job` — **the named flake** + +No new failures. + +Two diff-innocence checks, since counts drift and "trust the run" cuts both ways: +- The argless baseline first read **71**. Both extra errors were **mine**, in + `tests/unit/test_code_rebuild.py` (a generator fixture annotation and a `str | None` assignment). + Fixed → **69**, exactly the pinned baseline. My changes contribute zero. +- The self-containment ratchet reports **20 forbidden imports**, and **none are in a file I + touched** (`core/stores/vectorstore.py`, `core/stores/memberships.py`, + `core/typedshims/lancedb.py` do not appear). `core/ingest/code_corpus.py:87 → ops` is bp-152's + pre-existing reach. +- `test_scheduler_live` constructs its own `Supervisor` with `handlers={"ping": handler}` — it never + calls `build_components`, so the handler I registered cannot reach it. Its failure is an empty + response from a live model. + +--- + +### Write-scope amendment (made in-branch, commit `dbf6ae2`) + +Added **`ops/lifecycle/launcher.py`** (Item 5's entire target — the probe at `:374-393` and its call +site at `:551` — plus Item 7's `Launcher` method and handler registration) and +**`scheduler/code_sync.py`** (the job kind + checkpointing handler the verb enqueues, placed beside +its two siblings rather than in a new module that splits one lane across two homes). Bare globs, no +inline comments. Status field untouched. **Third instance this wave** → issue #52 proposes deriving +`write_scope` from the items' own `file:line` citations instead of restating it by hand. + +--- + +### In-flight + +Nothing. All seven items are built and the branch is gate-green. + +### Next action + +Owner reviews and merges the PR. Then, in order: `deploy` (which #39 has been holding) → +`palace code-rebuild --dry-run` to re-confirm the ratio at the then-current cut → +`palace code-rebuild` to enqueue the live run. + +### Open questions + +- **#50 — D4/F3 is falsified on the real ledger.** "Chain members are a strict subset of the version + set" is not true of the shipped reader: measured **1,663 versions, 1,663 chain members, an empty + difference both ways**. The reasoning assumed chains follow HEAD's first-parent line; they do not + — `capture_commit_diffs` is handed *every* snapshotted commit, each diffed against **its own** + first parent, so a side branch's own linear history is captured too. Nothing in the code relies on + the strict-subset reading and the rebuild is correct either way, but §4 quantifies invariants with + that clause and F3's disposition rests on it. A note amendment is not a builder's hand. +- **#51 — D7's L1 lane figure is 19% off the shipped chunkers** (2.42× noted, **1.96×** measured) + while the aggregate holds at −2.6%. The L1 *atom* count matches (3,482 vs 3,424); the *chunk* + count is 17.6% lower than the note's probe produced. Exactly the shape Amendment A1.4 warns about + — small enough to be invisible in the total, in the number the design uses to prove itself. +- **#52 — `write_scope` omits files the plan's own items cite by `file:line`**, third instance. +- **#18 stays OPEN** (the `code_sync` TimeoutError wedge). The slicing addresses its failure mode + and Item 2 demonstrates the bound, but it closes when the **live** capture succeeds post-deploy, + not when the machinery lands. +- **The derived board is stale by ~34 rows** (bp-128…bp-150, bp-154, the erratum-relation track). + `scripts/board.py --write` was run, produced 34 lines of unrelated catch-up, and was **reverted** + — that drift belongs to a `/triage` sweep, not to this PR. Flagged so it is not lost. + +### Context-manifest delta + +Read beyond §2, and worth a successor's time: +- `scheduler/queue.py:395-434, 545-560` — the `claim`/`checkpoint` lease protocol. Essential: this + plan is its first consumer, and the per-batch-vs-per-job deadline distinction is the whole reason + a long rebuild is safe. +- `scheduler/supervisor.py:280-297` — the dispatch seam. "Complete only if still RUNNING after the + handler returns" is what makes a self-checkpointing handler yield rather than finish. +- `docs/tracks/code-ingest.md` + `scripts/board.py` — the deskcheck queue is **generated**, so the + statement goes in the manifest's `backlog_deskcheck`, never by hand-editing the view. +- `tests/unit/test_memberships.py:545-572` — bp-152's `degenerate` fixture already carries all four + §8(f) shapes. Item 4's gauges reuse it rather than build a fifth. + +### Read-map (where a successor starts) + +1. `ops/code_rebuild.py` — the whole rebuild, module docstring first: it names the four movements + and, more usefully, says what the resume token **is not**. +2. `tests/unit/test_code_rebuild.py` — every acceptance with its degenerate input named in the + docstring and its precondition asserted first. +3. `docs/design-notes/vector-membership-store.md` D6/D7/§3/§6/§8(g) — the contract, read against + issues #50 and #51, which are where it stopped matching the code. + +### Follow-through + +- **Built?** Yes — all seven items. +- **Wired/delivered?** Yes. `palace code-rebuild` is listed, dispatched, its kind is registered on + the daemon, and `--dry-run` runs the read-only pass. The ON path exists, not merely the code + behind it. +- **Consumer?** The owner, via the verb; the daemon's catch-up probe, which now reads fibers; and + the standing `frequency_gauges`, which keep `|M|/|V|` — the D7 falsifier — observable forever + rather than measured once. +- **Track state?** The **vector-membership arc** (bp-151 → bp-155 → bp-152 → bp-153) is **READY TO + DESKCHECK** on this merge, filed into `docs/tracks/code-ingest.md`'s `backlog_deskcheck` (the + generated queue's actual source). **Not done, and not self-declared** — the arc's closing act is + the live rebuild, which is the owner's. The rest of the code-ingest track (bp-095, the seed run, + integrator densification) remains **work-owed**, not deskcheck-owed. +- **New findings?** #50 (F3 falsified by measurement), #51 (the L1 figure), #52 (`write_scope` + derivation). One defect found by an acceptance test and fixed in-branch: `pending_commits` called + `_ensure_schema` before reading — a `CREATE TABLE` that raises on a `mode=ro` connection and + silently mutates the file otherwise. The dry-run reads the LIVE ledger, where those tables have + never existed, so this was the ordinary path, not a corner. Now ratcheted. diff --git a/docs/build-plans/bp-153/plan.md b/docs/build-plans/bp-153/plan.md index e323f39..04e25d2 100644 --- a/docs/build-plans/bp-153/plan.md +++ b/docs/build-plans/bp-153/plan.md @@ -14,6 +14,8 @@ write_scope: - core/typedshims/lancedb.py - core/ingest/code_corpus.py - scripts/palace.py + - ops/lifecycle/launcher.py + - scheduler/code_sync.py - tests/unit/test_code_rebuild.py - tests/unit/test_code_lineage.py - tests/unit/test_memberships.py @@ -172,6 +174,18 @@ Production files: - `scripts/palace.py` — the owner-visible `code-rebuild` verb. (Note: `down`/`up`/`restart`/ `deploy` live here, **not** on the `mind-palace` wrapper.) +**Amended at build (2026-08-12, bp-153 session 1)** — two files the §7 items name but the +`write_scope` list omitted, added above so the reviewer's lane matches the work: + +- `ops/lifecycle/launcher.py` — Item 5's whole target. The probe being re-homed is + `_code_backfill_incomplete` (`:374-393`) with its one call site (`:551`); the file was never + listed. Item 7's verb also needs its `Launcher` method and its handler registration here. (Third + instance of this omission this wave — the graduation checklist should derive `write_scope` from + the items' cited `file:line`s rather than restating it.) +- `scheduler/code_sync.py` — Item 7's verb must ENQUEUE, and the job kind + checkpointing handler + belong beside their two siblings (`code_sync`, `code_backfill`) rather than in a new module that + splits one lane across two homes. + Test files carried: `tests/unit/test_code_rebuild.py` (new), `tests/unit/test_code_lineage.py` (pins chain behavior + the corrected docstring), `tests/unit/test_memberships.py` (gauges join the membership tests), diff --git a/docs/tracks/code-ingest.md b/docs/tracks/code-ingest.md index 3439aeb..ce18a65 100644 --- a/docs/tracks/code-ingest.md +++ b/docs/tracks/code-ingest.md @@ -13,7 +13,7 @@ dod: - CI-wiring the ENABLE path — CodeIngestConfig + daemon enqueue + `palace code-seed` (bp-098, warrant finding-0159) - the seed run PROVES it works — code is actually embedded + retrievable (not just built); owner-visible run - integrator densification (finding-0151) — design-pass, FABLE, after the build plans -backlog_deskcheck: null +backlog_deskcheck: "The vector-membership arc (dn-vector-membership-store; bp-151 → bp-155 → bp-152 → bp-153, all merged) is READY TO DESKCHECK on bp-153's merge. Demo: `palace code-rebuild --dry-run` reporting the measured factor at the live cut (2.280x over 1,663 versions: 52,200 chunks vs 22,897 atoms, 15,186-atom carry-forward seed); the standing `|M|/|V|` gauge reproducing it after a run; the re-homed incompleteness probe reading fibers instead of the shed `('','')` tuple. NOTE the live rebuild itself has NOT run — it is owner-gated behind the deploy (issue #39), so the deskcheck is of the machinery + the measurement, and the arc is not closed until the live run lands. The REST of the code-ingest track (bp-095, the seed run, integrator densification) remains work-owed, not deskcheck-owed." links: - docs/design-notes/code-ingest-pipeline.md - docs/findings/finding-0151.md @@ -38,3 +38,11 @@ working as expected" ([[deskcheck-discipline]]); this track cannot be deskchecke — bp-095 (gated on M-C4), bp-098 (the wiring), the seed run, and integrator densification remain. It becomes deskcheck-ready only once it demonstrably ingests code. Do NOT surface this as "deskcheck-owed" until then; surface it as work-owed. + +**One ARC inside the track is deskcheck-ready** (bp-153, 2026-08-12): the vector-membership store — +`dn-vector-membership-store`, plans bp-151 → bp-155 → bp-152 → bp-153 — is a complete three-plan +family whose machinery and measurement can be shown working. It is carried in +`backlog_deskcheck` above rather than by flipping the track's phase, precisely because the rest of +the track is still work-owed and a track-level flip would over-claim. ⚑ The arc's own closing act, +the LIVE rebuild, has not run: it is owner-gated behind the deploy (issue #39). So the deskcheck is +of the machinery + the measured baseline, and the arc is not closed until the live run lands. diff --git a/ops/code_lineage.py b/ops/code_lineage.py index 098c481..ac35da5 100644 --- a/ops/code_lineage.py +++ b/ops/code_lineage.py @@ -134,9 +134,23 @@ def supersession_chains(db: sqlite3.Connection) -> dict[str, list[str]]: """Per-path blob supersession chains `{path: [blob v0, v1, …]}` in ledger commit order (D4/D5). Threaded from `commit_diffs` walked in the `snapshots` capture order (rowid): a path's chain is - the ordered distinct sequence of its blobs — the initial `old_blob` (if the file pre-existed the - window) then each `new_blob` as the file evolves; a delete (`new_blob=''`) ends presence without - adding a version, and a rename's add starts the new path's own chain (PD-1). The result is plain + its blobs in commit order with only **ADJACENT** repeats collapsed — the initial `old_blob` (if + the file pre-existed the window) then each `new_blob` as the file evolves; a delete + (`new_blob=''`) ends presence without adding a version, and a rename's add starts the new + path's own chain (PD-1). + + [banner: correction] This said "the ordered **distinct** sequence of its blobs", which is not + what `:155` does and not what the design needs. The collapse is adjacent-only, so a revert + survives: A → B → A threads `[A, B, A]` — three runs, two edges — where a distinct-collapse + would thread `[A, B]` and erase the revert entirely. The distinction is load-bearing rather + than pedantic: dn-vector-membership-store's F4 dispute rests on exactly it (the file-grain half + of the panel's "the atom view is cyclic" finding is DISPUTED on this code), and §4's + per-slot `|edges| = |runs| - 1` is the same formulation at slot grain + (`core.stores.memberships.slot_runs`). A docstring contradicting the behavior its own design + cites as evidence is the issue #28 defect class in miniature. Verified against the real ledger + (bp-153 Item 2, first successful capture over 1,377 commits): `ops/lifecycle/launcher.py` + re-occupies one blob at NON-adjacent run positions, which a distinct-collapse would have lost. + The result is plain data for `core.kernel.temporal.boundary.poset_from_chains` (a chain = a total order; the corpus is the disjoint union of chains — a forest, §8). NB the poset core's contract is `dict[str, list[int]]` (version_seq, `acquire.py`) and it re-sorts its values, so a temporal diff --git a/ops/code_rebuild.py b/ops/code_rebuild.py new file mode 100644 index 0000000..00301a4 --- /dev/null +++ b/ops/code_rebuild.py @@ -0,0 +1,694 @@ +# ── Family: code rebuild — the one deliberate migration into the atom+membership model (bp-153) ── +# OBJECT: the D7 rebuild — a read-only baseline that re-derives the dedup economics at the +# CURRENT ledger cut, the step-0 `commit_diffs` capture, the carry-forward seed, and the +# sliced/checkpointed/resumable walk that lands every ledger version as a fiber. +# INVARIANT: the OLD duplicated backfill is never run from here (52,755 vs 22,502 embeds is 2.34× +# measured waste, D7). Every slice leaves DURABLE progress before it yields, so a run +# killed mid-slice resumes without double-landing — and correctness does not depend on +# the resume token at all: derivation is pure, so a re-landed fiber is equal by +# construction (D2 step 3) and reconciliation converges (D2 step 4). +# ENFORCED: structural — the rebuild NEVER stops the daemon and never writes from the CLI process: +# it runs as a BACKGROUND queue job under the single writer (D7/S3), and the read-only +# baseline holds no write handle to any store it measures. +"""The rebuild (dn-vector-membership-store D7/§3, bp-153; warrant finding-0168). + +The membership store (bp-152) gave the corpus a lander; this module is the one deliberate migration +that fills it. Four movements, in strict blast-radius order — the order IS the design, because each +one is the precondition of the next: + + * **`measure_baseline` — read-only, and it writes nothing.** The economics that justified this + rebuild were measured on the 2026-07-27 ledger cut; the corpus grows, so the constant is not + portable and the RATIO is. This re-derives Σ per-version chunks (the duplicated model), the + distinct atom count under D0, their ratio, the carry-forward seed, and whether that seed's + embedder identity still matches the live config — at the cut the rebuild is actually about to + run against. Nothing downstream may spend an embed before this has reported (§8 g). + * **`capture_slice` — step 0, the first successful `commit_diffs` capture.** The machinery has + shipped since bp-099 and has NEVER successfully run (the one attempt, job 300240, died in + `TimeoutError` on 2026-07-25). Its resume token is not a token at all: `_commit_diffs_captured` + is a durable per-commit marker, so the budget is enforced BETWEEN commits and every commit that + completed stays completed. That is what makes a time-budgeted slice safe (§3 Q2). + * **`seed_carry_forward` — the bulk embed-reuse, at zero embed cost.** The live store already + holds L0a/L0b rows whose CANONICAL (header-free) body is exactly the atom D0 now identifies. The + seed re-keys those rows to their atom id and COPIES the stored vector — no embedder call. It is + governed by the embedder pin (bp-152 §6): a vector from another geometry is not a reuse, it is a + silent corruption of the one ANN space, so the seed refuses unless the identity matches. L1 is + deliberately excluded — its stored windows were cut over header-bearing prose and do not survive + D0's windowing pin, so L1 re-embeds (D7). + * **`rebuild_slice` — the walk.** Every ledger version becomes a fiber (D4/F3: a side-branch + version sits on no chain but is still a member of M), landed through the D2 write path with its + HEAD blob passed explicitly so currency reconciles correctly for a non-HEAD version. + +**What the resume token is and is not.** The token records POSITION so a resumed run does not +re-derive work already done; it is not what makes the run safe. Landing version `v` twice is a +no-op by construction — the fiber is derived purely from `(path, source)` and `INSERT OR IGNORE` +leaves the existing rows standing — so a crash between the fiber write and the checkpoint costs one +re-derivation and changes nothing. R6's falsifier is therefore narrow and real: a slice that +exceeds its budget WITHOUT leaving durable progress. Every slice here leaves it before it yields. +""" + +from __future__ import annotations + +import json +import sqlite3 +import time +from collections.abc import Callable, Sequence +from dataclasses import dataclass, field +from datetime import timedelta +from hashlib import sha256 +from pathlib import Path +from typing import Any + +from core.ingest.code_corpus import CodeCorpusSync, atom_id, derive_code_chunks +from core.kernel.provenance import Provenance +from core.stores.memberships import EmbedderIdentity, MembershipStore +from core.stores.vectorstore import ( + ATOM_ROW_SHED, + LAYER_CODE_AST, + LAYER_CODE_TEXT, + VectorStore, +) +from ops.code_lineage import capture_commit_diffs, ledger_commits, ledger_versions +from ops.code_snapshot import read_py_blobs + +# How many ledger versions one slice derives before the clock is consulted. Small enough that a +# budget is honored promptly, large enough that the per-slice git batch read stays worthwhile. +VERSION_SLICE = 64 + +# How many distinct blobs one `git cat-file --batch` reads. Bounds peak memory on the read, never +# the semantics — a slice's versions are the unit of progress, not the blob batch. +BLOB_BATCH = 256 + +# The layers whose STORED embed text still recovers its canonical body (D7's carry-forward seed). +# L1 is absent deliberately, and that absence is the measured half of D0: its windows were cut over +# header-bearing prose, so a stored L1 window is not a window the current chunker would ever emit. +CARRY_FORWARD_LAYERS = (LAYER_CODE_AST, LAYER_CODE_TEXT) + + +# ── recovering an atom's identity from a row the OLD model wrote ───────────────────────────── + + +def canonical_body_of_stored(layer: str, text: str, source_path: str) -> str | None: + """The CANONICAL (header-free) body behind a pre-D1 stored embed rendering, or None if this row + cannot carry forward. + + ⚑ A wrong strip is silent identity corruption — it mints an atom nobody can ever hit again — so + the header is not guessed from the shape of the line. L0a's stored text is + `# {source_path}:{qualname}{signature}\\n{body}`, and the row still carries `source_path`, so + the first line is verified to be THAT path's header before anything is removed. A body line that + legitimately begins with `#` cannot pass that check. + + * **L0a** (`code_ast`) — strip the verified header line; the remainder is the identity input. + * **L0b** (`code_text`) — raw-source windows are headerless by construction, so the stored text + IS its own canonical body. + * **L1** (`codedoc`) — None, always. Under D0 the windows are cut over canonical prose while + these were cut over header-bearing prose, so the stored window is not a chunk the current + derivation emits; carrying it forward would land geometry for an atom that does not exist + (D7: L1 recuts and re-embeds). + """ + if layer == LAYER_CODE_TEXT: + return text + if layer != LAYER_CODE_AST: + return None + head, sep, body = text.partition("\n") + if not sep or not head.startswith(f"# {source_path}"): + return None + return body + + +def atom_id_of(layer: str, canonical_body: str) -> str: + """`"{layer}:{content_hash}"` from a canonical body — the same identity `atom_id` mints from a + live `CodeChunk`, spelled once here so a carried-forward row and a freshly derived chunk can + never disagree about what an atom is (D1).""" + return f"{layer}:{sha256(canonical_body.encode('utf-8')).hexdigest()}" + + +# ── Item 1: the read-only baseline ─────────────────────────────────────────────────────────── + + +@dataclass(frozen=True) +class LayerBaseline: + """One lane's half of the measurement. `ratio` is the lane's dedup factor at this cut.""" + + chunks: int = 0 + atoms: int = 0 + seed: int = 0 + + @property + def ratio(self) -> float: + return (self.chunks / self.atoms) if self.atoms else 0.0 + + +@dataclass(frozen=True) +class EmbedderCheck: + """Whether the carry-forward seed's geometry is the one the live config would produce. + + The check is deliberately split into what the store CAN answer and what it structurally cannot. + `dim` is recoverable from any stored vector's length. **`model` is not recorded on a pre-D1 + row** — the vector table's Arrow schema is shared with the prose lane and has no embedder + column, which is precisely the gap bp-152's `atoms` ledger exists to close. So for rows the + membership ledger already knows, the model is checked against the recorded identity; for the + legacy rows the honest answer is "unrecorded", and `unrecorded_rows` says how many carry that + status rather than letting a silent assumption ride into a 13k-atom bulk reuse.""" + + live: EmbedderIdentity + stored_dim: int | None = None + ledger_atoms: int = 0 + ledger_mismatched: int = 0 + unrecorded_rows: int = 0 + + @property + def dim_match(self) -> bool: + return self.stored_dim is None or self.stored_dim == self.live.dim + + @property + def matches(self) -> bool: + """True when nothing CHECKABLE contradicts reuse. An unrecorded model is not a match claim + — `unrecorded_rows` carries that, and the caller decides whether the evidence outside the + store (the config's own history) closes it.""" + return self.dim_match and self.ledger_mismatched == 0 + + +@dataclass(frozen=True) +class Baseline: + """The rebuild's economics, re-derived at the CURRENT cut (§3 Q6/Q7, §8 g). + + `ratio` is the portable claim; the absolute counts are not, because the corpus grows between + the measurement and the run. The design's falsifier compares THIS ratio to 2.34× like-for-like, + and a deviation beyond ~10% is a finding rather than a shrug.""" + + versions: int = 0 + unreadable: int = 0 + chunks: int = 0 + atoms: int = 0 + seed: int = 0 + per_layer: dict[str, LayerBaseline] = field(default_factory=dict) + embedder: EmbedderCheck | None = None + + @property + def ratio(self) -> float: + """Σ per-version chunks ÷ distinct atoms — the dedup factor the rebuild is buying.""" + return (self.chunks / self.atoms) if self.atoms else 0.0 + + @property + def embeds_avoided(self) -> int: + """Embeds the membership model does not pay that the duplicated model would have.""" + return self.chunks - self.atoms + + def __str__(self) -> str: + lanes = " · ".join(f"{k} {v.ratio:.2f}×" for k, v in sorted(self.per_layer.items())) + return (f"versions={self.versions} chunks={self.chunks} atoms={self.atoms} " + f"ratio={self.ratio:.2f}× ({lanes}) seed={self.seed} " + f"embeds_avoided={self.embeds_avoided}") + + +def _slice_sources(repo: Path, versions: Sequence[tuple[str, str]]) -> dict[str, str]: + """blob_sha -> source for one slice, in ONE batched git read (φ_code's own reader, never a + second one). A blob git cannot produce (shallow clone, pruned object) is simply absent.""" + return read_py_blobs(repo, sorted({b for _, b in versions})) + + +def measure_baseline(sync: CodeCorpusSync, db: sqlite3.Connection, *, + progress: Callable[[int, int], None] | None = None) -> Baseline: + """Re-derive the rebuild's target and cost at the current ledger cut. **Writes nothing.** + + The two counts are the whole point and they are computed over the SAME version set, in one + walk, so they cannot drift apart: + + * Σ per-version chunks — what the duplicated model would embed, one row per chunk per version; + * |distinct atoms| — what the membership model embeds, one row per `(layer, content_hash)` + under D0's header-free identity, corpus-wide. + + The version set is `ledger_versions` — every distinct `(path, blob_sha)` the ledger recorded, + not only chain members: a side-branch version lands a fiber though it sits on no chain (D4/F3), + so quantifying over chains here would under-count the rebuild's work. + + No figure this function reports is a constant. That is not tidiness: the same defect class + (issue #28) has docstrings in this repo claiming an edge count 8.4× off the live store, and a + baseline that inherited July's 22,502 would authorize spending against a number nobody + re-checked.""" + versions = ledger_versions(db) + chunks = 0 + atoms: set[str] = set() + per_layer_chunks: dict[str, int] = {} + per_layer_atoms: dict[str, set[str]] = {} + unreadable = 0 + + for start in range(0, len(versions), BLOB_BATCH): + window = versions[start:start + BLOB_BATCH] + sources = _slice_sources(sync.repo, window) + for path, blob_sha in window: + source = sources.get(blob_sha) + if source is None: + unreadable += 1 + continue + for ch in derive_code_chunks(path, source, max_chars=sync.max_chars, + overlap_chars=sync.overlap_chars): + chunks += 1 + cid = atom_id(ch) + atoms.add(cid) + per_layer_chunks[ch.layer] = per_layer_chunks.get(ch.layer, 0) + 1 + per_layer_atoms.setdefault(ch.layer, set()).add(cid) + if progress is not None: + progress(min(start + BLOB_BATCH, len(versions)), len(versions)) + + seed_ids = carry_forward_candidates(sync.store, targets=atoms) + embedder = _check_embedder(sync.store, sync.memberships, sync.embedder_identity, seed_ids) + per_layer = { + layer: LayerBaseline( + chunks=per_layer_chunks.get(layer, 0), + atoms=len(per_layer_atoms.get(layer, ())), + seed=sum(1 for cid in seed_ids if cid.startswith(f"{layer}:")), + ) + for layer in sorted(set(per_layer_chunks) | set(per_layer_atoms)) + } + return Baseline(versions=len(versions), unreadable=unreadable, chunks=chunks, + atoms=len(atoms), seed=len(seed_ids), per_layer=per_layer, + embedder=embedder) + + +def _check_embedder(vectors: VectorStore, memberships: MembershipStore, + live: EmbedderIdentity, seed_ids: set[str]) -> EmbedderCheck: + """Is the seed's stored geometry the live one? Read-only, and honest about what it cannot see. + + Reuse across a model change puts two geometries in ONE ANN space, and no downstream measurement + can detect it — which is why the owner pinned reuse to the embedder identity and why this runs + BEFORE Item 3 spends anything.""" + sample = vectors.project(["vector"], where=f"provenance = '{Provenance.CODE.value}'", limit=1) + stored_dim = len(list(sample[0]["vector"])) if sample and sample[0].get("vector") else None + known = memberships.known_atoms(seed_ids, live) if seed_ids else set() + ledger_all = memberships.ledger_atom_ids() & seed_ids if seed_ids else set() + return EmbedderCheck(live=live, stored_dim=stored_dim, ledger_atoms=len(ledger_all), + ledger_mismatched=len(ledger_all - known), + unrecorded_rows=len(seed_ids - ledger_all)) + + +# ── the carry-forward seed: bulk embed-reuse by canonical re-hash ──────────────────────────── + + +def carry_forward_candidates(vectors: VectorStore, *, + targets: set[str] | None = None) -> set[str]: + """Atom ids recoverable from rows the OLD model already embedded — the D7 seed, as a set. + + Read-only and vector-free: only `id`, `layer`, `text` and `source_path` are projected, so + measuring the seed never pays for 2560 floats per row. `targets` restricts the answer to atoms + the CURRENT derivation actually emits, which is what makes the count meaningful — a stored row + cut under a superseded rule (bp-151's A1.2 threshold change, say) re-hashes to an atom no + version would ever reference, and counting it would inflate the seed with dead geometry.""" + seed: set[str] = set() + for row in vectors.project(["id", "layer", "text", "source_path"], + where=f"provenance = '{Provenance.CODE.value}'"): + layer = str(row.get("layer") or "") + if layer not in CARRY_FORWARD_LAYERS: + continue + source_path = str(row.get("source_path") or "") + if not source_path: # already a shed atom row (D1) — not a candidate + continue + body = canonical_body_of_stored(layer, str(row.get("text") or ""), source_path) + if body is None: + continue + cid = atom_id_of(layer, body) + if targets is None or cid in targets: + seed.add(cid) + return seed + + +@dataclass(frozen=True) +class SeedReport: + """What the carry-forward actually moved. `embeds` is asserted zero by the acceptance test — + the seed's entire claim is that ~59% of the atom set enters the plane without an embedder + call.""" + + rows_written: int = 0 + atoms_recorded: int = 0 + embeds: int = 0 + refused_embedder_mismatch: bool = False + + +def seed_carry_forward(vectors: VectorStore, memberships: MembershipStore, + embedder: EmbedderIdentity, *, targets: set[str], + stored_dim: int | None = None) -> SeedReport: + """Land the seed atoms by COPYING stored vectors onto atom-keyed rows (D7). Zero embeds. + + Each pre-D1 row whose canonical body hashes to a wanted atom becomes a new shed atom row + (`ATOM_ROW_SHED`) carrying the SAME vector, and is recorded in the membership ledger under + `embedder` so `land()` sees it as present and never re-embeds it. The old row is neither + deleted nor modified — append-only holds, and retiring the duplicated model is a separate, + explicitly reported step (`retire_legacy_rows`). + + **The pin is a refusal, not a warning.** If the stored dimension contradicts the live config + the seed does nothing at all and says so: a partial seed under a changed embedder would put two + geometries in one ANN space, which is the one failure this whole path exists to prevent. + + `current=False` on every seeded row is correct rather than conservative: at seed time the atom + has no occupancy — the fibers land afterwards — and `land()` raises `current_any` in step 5 for + exactly the atoms whose current-membership count crosses 0→1 (D8's write order).""" + if stored_dim is not None and stored_dim != embedder.dim: + return SeedReport(refused_embedder_mismatch=True) + already = memberships.known_atoms(targets, embedder) + wanted = targets - already + rows: dict[str, dict[str, Any]] = {} + for row in vectors.project(["id", "layer", "text", "source_path", "vector"], + where=f"provenance = '{Provenance.CODE.value}'"): + layer = str(row.get("layer") or "") + source_path = str(row.get("source_path") or "") + if layer not in CARRY_FORWARD_LAYERS or not source_path: + continue + text = str(row.get("text") or "") + body = canonical_body_of_stored(layer, text, source_path) + if body is None: + continue + cid = atom_id_of(layer, body) + if cid not in wanted or cid in rows: + continue + rows[cid] = { + **ATOM_ROW_SHED, + "id": cid, "title": "", "provenance": Provenance.CODE.value, + "text": text, "layer": layer, "current": False, + "vector": [float(v) for v in list(row["vector"])], # copied, never re-embedded + } + if not rows: + return SeedReport() + written = vectors.add(list(rows.values())) + recorded = memberships.record_atoms( + [(cid, str(r["layer"])) for cid, r in rows.items()], embedder) + return SeedReport(rows_written=written, atoms_recorded=recorded, embeds=0) + + +# ── Item 2: step 0 — the sliced `commit_diffs` capture ─────────────────────────────────────── + + +@dataclass(frozen=True) +class CaptureProgress: + """One capture slice's outcome. `remaining > 0` with `budget_spent` true is the NORMAL yield — + the slice ran out of clock and stopped at a commit boundary, having left every commit it + finished durably marked.""" + + captured: int = 0 + remaining: int = 0 + budget_spent: bool = False + + @property + def done(self) -> bool: + return self.remaining == 0 + + +def pending_commits(db: sqlite3.Connection) -> list[str]: + """Ledger commits whose diffs are not yet captured, in capture order. Empty ⇒ step 0 is done. + + Reading the marker table IS reading the resume position, which is why this lane needs no + separate token: progress lives where the work landed. + + ⚑ **This is a READ, so it does not create the schema it reads.** The obvious spelling calls + `_ensure_schema` first (the tables may not exist — on the live ledger they never have), and + that is a write: a CREATE TABLE on a connection the caller opened `mode=ro` raises, and on a + read-write connection it silently mutates the file a dry-run promised not to touch. The + read-only baseline behind `palace code-rebuild --dry-run` calls this against the LIVE ledger, + so the absent-table case is answered instead of created: no marker table means nothing has + been captured, which is the true answer and the one step 0 exists to act on.""" + marked: set[str] = set() + exists = db.execute( + "SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = '_commit_diffs_captured'" + ).fetchone() + if exists: + marked = {str(s) for (s,) in db.execute("SELECT commit_sha FROM _commit_diffs_captured")} + return [c for c in ledger_commits(db) if c not in marked] + + +def capture_slice(db: sqlite3.Connection, repo: Path, *, budget_s: float = 60.0, + batch: int = 25) -> CaptureProgress: + """Capture `commit_diffs` for as many ledger commits as `budget_s` allows, then yield (D7/R6). + + The shipped `capture_commit_diffs` has never successfully run against real data — the one + attempt died in `TimeoutError` (job 300240, 2026-07-25) because the whole history rode one + unbounded job. This wraps it in a clock without changing what it does: commits are handed over + in small batches and the budget is checked BETWEEN batches, never inside one. + + **Why that is safe, and why it satisfies R6.** Each commit is captured under its own `with db:` + transaction and marked in `_commit_diffs_captured` (`ops/code_lineage.py:119-129`). So the + durable progress is the marker table itself: a slice that stops after batch *k* has committed + batches 1..*k*, and the next slice re-computes `pending_commits` and continues. There is no + window in which work is done but unrecorded, which is exactly the state R6's falsifier names — + "a slice exceeding its budget WITHOUT leaving a checkpoint". A budget overrun here leaves one. + + The first batch always runs, whatever the budget: a slice that yields having done nothing would + turn a tight budget into a livelock rather than slow progress.""" + pending = pending_commits(db) + if not pending: + return CaptureProgress() + deadline = time.monotonic() + budget_s + captured = 0 + index = 0 + while index < len(pending): + window = pending[index:index + batch] + captured += capture_commit_diffs(db, repo, window) + index += len(window) + if time.monotonic() >= deadline and index < len(pending): + return CaptureProgress(captured=captured, remaining=len(pending) - index, + budget_spent=True) + return CaptureProgress(captured=captured, remaining=0) + + +# ── Item 3: the sliced, checkpointed, resumable rebuild ────────────────────────────────────── + + +@dataclass(frozen=True) +class RebuildProgress: + """One rebuild slice's outcome, and the resume token that positions the next one. + + `token` is JSON so the queue's `checkpoint` column carries it unchanged; `None` means the walk + finished. `atoms_embedded` is the number that must stay near zero for the carry-forward lanes — + it is the acceptance's arithmetic, not a log line.""" + + versions_landed: int = 0 + atoms_embedded: int = 0 + atoms_reused: int = 0 + membership_rows: int = 0 + remaining: int = 0 + token: str | None = None + + @property + def done(self) -> bool: + return self.token is None + + +def _resume_index(token: str | None, versions: Sequence[tuple[str, str]]) -> int: + """Where a resumed run picks up. A token naming a version resumes AT it (never after), so a + version whose fiber write and whose checkpoint straddled a crash is simply re-landed — which is + a no-op, because derivation is pure and `write_fiber` leaves an existing fiber standing.""" + if not token: + return 0 + try: + mark = json.loads(token) + path, blob = str(mark["path"]), str(mark["blob_sha"]) + except (ValueError, KeyError, TypeError): + return 0 # an unreadable token costs a re-walk, never a + for i, (p, b) in enumerate(versions): # wrong landing — the walk is idempotent + if (p, b) == (path, blob): + return i + return 0 + + +def rebuild_slice(sync: CodeCorpusSync, db: sqlite3.Connection, *, token: str | None = None, + budget_s: float = 120.0, slice_size: int = VERSION_SLICE) -> RebuildProgress: + """Land one time-budgeted slice of the ledger's versions, then yield with a resume token. + + Every distinct `(path, blob_sha)` the ledger recorded becomes a fiber — not only the chain + members, because a side-branch version lands a fiber while sitting on no chain (D4/F3). Each + version is landed through the D2 write path with its path's HEAD blob passed explicitly, so a + historical (non-HEAD) version reconciles to `current=false` and the HEAD one to `current=true` + rather than the last-landed version winning. + + **Resumability is a property of the walk, not of the token.** Landing a version twice writes no + row (`INSERT OR IGNORE` over a purely derived fiber) and embeds nothing (the atoms are already + in the plane), and reconciliation converges on every call — so the token only saves + re-derivation time. The acceptance test kills a run mid-slice and resumes precisely to + demonstrate that the invariant does not depend on the token surviving. + + **The old duplicated backfill is never called from here.** `CodeCorpusSync.backfill` would walk + the same versions, but this walk exists to be SLICED and budgeted; the invariant that matters is + the one D7 states as measured waste, and it is honored by landing atoms, not by which function + is called.""" + versions = ledger_versions(db) + start = _resume_index(token, versions) + head = dict(_head_blobs(sync)) + lander = sync.lander + deadline = time.monotonic() + budget_s + landed = embedded = reused = rows = 0 + index = start + + while index < len(versions): + window = versions[index:index + slice_size] + sources = _slice_sources(sync.repo, window) + for path, blob_sha in window: + source = sources.get(blob_sha) + index += 1 + if source is None: # blob unreachable (shallow/pruned) — skip, said + continue + chunks = derive_code_chunks(path, source, max_chars=sync.max_chars, + overlap_chars=sync.overlap_chars) + if not chunks: + continue + report = lander.land(path, blob_sha, chunks, + head_blob_sha=head.get(path, blob_sha)) + landed += 1 + embedded += report.atoms_embedded + reused += report.atoms_reused + rows += report.membership_rows + if time.monotonic() >= deadline and index < len(versions): + nxt = versions[index] + return RebuildProgress( + versions_landed=landed, atoms_embedded=embedded, atoms_reused=reused, + membership_rows=rows, remaining=len(versions) - index, + token=json.dumps({"path": nxt[0], "blob_sha": nxt[1]})) + return RebuildProgress(versions_landed=landed, atoms_embedded=embedded, atoms_reused=reused, + membership_rows=rows, remaining=0, token=None) + + +def _head_blobs(sync: CodeCorpusSync) -> list[tuple[str, str]]: + """`(path, blob_sha)` at HEAD — reused from φ_code's blob walk, never a second git shell.""" + from ops.code_snapshot import list_py_blobs + + return list_py_blobs(sync.repo, "HEAD") + + +# ── the phase machine: what one queue slice of the rebuild does ────────────────────────────── + +PHASE_CAPTURE = "capture" # step 0 — `commit_diffs` (§3) +PHASE_SEED = "seed" # the carry-forward bulk reuse (D7) +PHASE_LAND = "land" # the sliced walk over every ledger version +PHASE_COMPACT = "compact" # physical maintenance + retiring the duplicated rows (§3) + + +@dataclass(frozen=True) +class StepResult: + """One queue slice's outcome. `token is None` means the whole rebuild is complete — which is + what the handler reads to decide between completing the job and checkpointing it.""" + + phase: str + message: str + token: str | None = None + + @property + def done(self) -> bool: + return self.token is None + + +def rebuild_step(sync: CodeCorpusSync, db: sqlite3.Connection, *, token: str | None = None, + capture_budget_s: float = 60.0, rebuild_budget_s: float = 120.0, + capture_batch: int = 25, slice_size: int = VERSION_SLICE) -> StepResult: + """Run ONE time-budgeted slice of the rebuild and return the token that positions the next. + + The phases are the D7/§3 ordering, and they are ordered because each is the other's + precondition: `capture` (step 0 — the chains' substrate, never successfully run before this + plan) → `seed` (bulk embed-reuse, so the walk pays for ~a third of the atoms rather than all of + them) → `land` (every ledger version becomes a fiber) → `compact` (the physical maintenance §3 + makes part of the store's semantics, plus retiring the duplicated rows the rebuild replaced). + + **No phase depends on the token for correctness.** `capture` resumes from its own marker table; + `seed` is keyed on atoms already present, so re-running it seeds nothing; `land` re-lands + idempotently; `compact` is physical. The token buys time, and only time — which is why killing + a run mid-slice and resuming is a test that passes rather than a risk that is managed. + + **The daemon is never stopped.** This is a queue citizen: it runs inside one job's dispatch, + yields at a slice boundary, and re-queues itself (D7/S3). The job class it enlarges is exactly + the one that wedged, which is why the budget is enforced and why every yield leaves durable + progress. + + `capture_batch` / `slice_size` are the units the budget is checked BETWEEN, exposed rather than + fixed because they set the granularity of yielding: a budget cannot interrupt a batch, so the + smallest slice that can be observed to yield is one batch. They are what the acceptance test + turns down to 1 to force mid-phase yields on a small fixture.""" + mark = _read_token(token) + phase = str(mark.get("phase") or PHASE_CAPTURE) + + if phase == PHASE_CAPTURE: + prog = capture_slice(db, sync.repo, budget_s=capture_budget_s, batch=capture_batch) + if not prog.done: + return StepResult(PHASE_CAPTURE, + f"capture: +{prog.captured} commits, {prog.remaining} to go", + json.dumps({"phase": PHASE_CAPTURE})) + return StepResult(PHASE_CAPTURE, f"capture: +{prog.captured} commits, COMPLETE", + json.dumps({"phase": PHASE_SEED})) + + if phase == PHASE_SEED: + # ONE derivation pass feeds all three: the target set, the seed's filter, and the pin's + # check. Deriving twice here would double the phase's cost for no new information. + seed_ids = carry_forward_candidates(sync.store, targets=_target_atoms(sync, db)) + check = _check_embedder(sync.store, sync.memberships, sync.embedder_identity, seed_ids) + report = seed_carry_forward(sync.store, sync.memberships, sync.embedder_identity, + targets=seed_ids, stored_dim=check.stored_dim) + if report.refused_embedder_mismatch: + # Not a silent downgrade: the seed is the geometry-mixing hazard the pin exists for, so + # a mismatch stops the phase rather than proceeding to re-embed 15k atoms unannounced. + return StepResult(PHASE_SEED, + "seed REFUSED: stored vector dim does not match the live embedder — " + "the carry-forward is not free and the rebuild's economics changed", + None) + return StepResult(PHASE_SEED, + f"seed: {report.rows_written} atoms carried forward at " + f"{report.embeds} embeds (of {len(seed_ids)} recoverable)", + json.dumps({"phase": PHASE_LAND})) + + if phase == PHASE_LAND: + walk = rebuild_slice(sync, db, token=token, budget_s=rebuild_budget_s, + slice_size=slice_size) + msg = (f"land: {walk.versions_landed} versions, +{walk.atoms_embedded} embeds, " + f"{walk.atoms_reused} reused, +{walk.membership_rows} occupancies") + if walk.token is not None: + nxt = dict(json.loads(walk.token)) + nxt["phase"] = PHASE_LAND + return StepResult(PHASE_LAND, f"{msg}, {walk.remaining} versions to go", + json.dumps(nxt)) + return StepResult(PHASE_LAND, f"{msg}, COMPLETE", + json.dumps({"phase": PHASE_COMPACT})) + + retired = retire_legacy_rows(sync.store) + compaction = sync.store.compact(older_than=timedelta(0)) + return StepResult(PHASE_COMPACT, + f"compact: {compaction}; {retired} legacy rows superseded (retained)", None) + + +def _read_token(token: str | None) -> dict[str, object]: + """A token is advisory, so an unreadable one restarts the walk rather than failing the job — + every phase is idempotent, so the cost of a bad token is time, never a wrong landing.""" + if not token: + return {} + try: + mark = json.loads(token) + except ValueError: + return {} + return dict(mark) if isinstance(mark, dict) else {} + + +def _target_atoms(sync: CodeCorpusSync, db: sqlite3.Connection) -> set[str]: + """Every atom the CURRENT derivation emits over the whole ledger — the seed's filter, so a row + cut under a superseded rule never carries geometry forward for an atom no version references.""" + targets: set[str] = set() + versions = ledger_versions(db) + for start in range(0, len(versions), BLOB_BATCH): + window = versions[start:start + BLOB_BATCH] + sources = _slice_sources(sync.repo, window) + for path, blob_sha in window: + source = sources.get(blob_sha) + if source is None: + continue + for ch in derive_code_chunks(path, source, max_chars=sync.max_chars, + overlap_chars=sync.overlap_chars): + targets.add(atom_id(ch)) + return targets + + +def retire_legacy_rows(vectors: VectorStore) -> int: + """Flip every pre-D1 duplicated CODE row to `current=false`, RETAINING it. Returns rows flipped. + + [cross-ref: extension] The plan does not enumerate this step and it is reported separately for + exactly that reason. The rebuild lands the atom plane beside the duplicated rows it replaces, + and both would answer the default current-view search — the old model's per-version rows are + what D1 supersedes, so leaving them current means the rebuild bought its dedup and then served + the duplication anyway. + + It is a SUPERSESSION, not a removal, and that is the whole reason it is expressible here: this + is keep-and-link (D2) applied to the rows the row model retired — nothing is deleted, `|V|` + does not decrease, and D5's "purge is the ONE removal" is untouched. A reviewer who disagrees + reverses it with one `update`, and the rows are all still there to reverse it with.""" + return vectors.supersede_legacy_code_rows() diff --git a/ops/lifecycle/launcher.py b/ops/lifecycle/launcher.py index d286464..ac06463 100644 --- a/ops/lifecycle/launcher.py +++ b/ops/lifecycle/launcher.py @@ -374,17 +374,32 @@ class Components: def _code_backfill_incomplete(cfg: Config, code_driver: CodeCorpusSync) -> bool: """The catch-up incompleteness probe (dn-temporal-code-corpus §3, bp-099): is the store missing any ledger code version? Compares DISTINCT `(path, blob_sha)` versions on BOTH sides — the - store's embedded code versions vs the ledger's `ledger_versions` — so a COMPLETE store is - exactly equal and the probe enqueues NOTHING (no loop). (§6's shorthand `distinct digests < - distinct versions` would false-positive forever — 1,472 distinct blobs < 1,542 distinct - (path,blob) pairs even when complete; the falsifier forbids that loop, so the probe is - like-to-like — finding-0166.) Cheap scans, no embed. A missing ledger → not incomplete.""" - from core.kernel.provenance import Provenance + versions the store holds vs the ledger's `ledger_versions` — so a COMPLETE store is exactly + equal and the probe enqueues NOTHING (no loop). (§6's shorthand `distinct digests < distinct + versions` would false-positive forever — 1,472 distinct blobs < 1,542 distinct (path,blob) + pairs even when complete; the falsifier forbids that loop, so the probe is like-to-like — + finding-0166.) Cheap scans, no embed. A missing ledger → not incomplete. + + [banner: correction] The store side read the vector rows' `(source_path, digest)` pairs. Under + dn-vector-membership-store D1 a code ATOM row carries neither column — both are shed to `''` + (`ATOM_ROW_SHED`) — so against a rebuilt store every atom row collapses to the SINGLE tuple + `('', '')` and the probe reads `1 < 1,663`: true forever, enqueueing a backfill on every daemon + start, for ever. That is finding-0166's named falsifier reappearing through a different door, + and bp-152 shipped the shed knowing this re-home was bp-153's to make. + + **The same number, a sturdier home** (the note's §6 re-home (1)): occupancy is the membership + relation's now, and a version IS its fiber `M(path, blob_sha)`, so the store-side count is + `memberships.fibers()` — distinct `(path, blob_sha)` pairs, which is precisely what the old + read was approximating with columns that happened to hold those two values. It is also the + honest reading of the F6 re-home: a version is present iff its fiber is non-empty. + + ⚑ **The backfill TRIGGERS are unchanged** and must stay so (§6): the cadence, the call site, + the `cfg.ingestion.code.enabled` gate and the enqueued job kind are all exactly as they were. + Only the probe's DATA SOURCE moved. Any trigger-level change here would be out of design.""" from ops.code_lineage import ledger_versions from ops.code_snapshot import open_snapshot_db - store_versions = {(str(r["source_path"]), str(r["digest"])) - for r in code_driver.store.all_rows(provenances={Provenance.CODE})} + store_versions = set(code_driver.memberships.fibers()) db = open_snapshot_db(cfg.paths.data_dir / "code_snapshots.sqlite") try: ledger = set(ledger_versions(db)) @@ -420,8 +435,10 @@ def build_components(cfg: Config) -> Components: ) from scheduler.code_sync import ( CODE_BACKFILL_KIND, + CODE_REBUILD_KIND, CODE_SYNC_KIND, code_backfill_handler, + code_rebuild_handler, code_sync_handler, enqueue_code_backfill, enqueue_code_sync, @@ -503,6 +520,12 @@ def health_check() -> list[Flag]: # Registered unconditionally (same species as code_sync); ENQUEUED only by the catch-up # incompleteness probe or the deliberate `palace code-backfill`. Idempotent, BACKGROUND. CODE_BACKFILL_KIND: code_backfill_handler(code_driver, code_snapshots_db, code_driver.repo), + # The D7 rebuild (bp-153 / dn-vector-membership-store): the ONE deliberate migration into + # the atom+membership model, run as CHECKPOINTED slices with a per-slice time budget so the + # lane that wedged on 2026-07-25 cannot wedge on this. Registered unconditionally (same + # species again); ENQUEUED only by the deliberate `palace code-rebuild` — never by a probe, + # because a rebuild is an owner-visible act, not a catch-up. + CODE_REBUILD_KIND: code_rebuild_handler(code_driver, code_snapshots_db, queue), # The L1 action-log projector (bp-069 Item 3): the sensor's DELAYED rate, model-less like # chat_sync. Re-extracts WHAT was performed (typed events, structural refs) from the raw # transcripts at housekeeping cadence, incrementally by transcript_digest. @@ -1061,6 +1084,66 @@ def code_backfill(self) -> int: print(f"code backfill: enqueued code_backfill job #{job.id}; {where}") return 0 + # --- code-rebuild (the D7 atom+membership migration, bp-153) ------------------------------ + def code_rebuild(self, *, dry_run: bool = False) -> int: + """The owner-visible enable act for dn-vector-membership-store D7. + + `--dry-run` runs the READ-ONLY baseline in this process and prints it — Σ per-version + chunks against distinct atoms at the CURRENT ledger cut, the ratio, the carry-forward seed, + and whether that seed's embedder identity still matches the live config. It writes nothing, + which is the point: the design's economics were measured on a July cut and the corpus grows, + so the ratio is the portable claim and this is where it gets re-derived before anything is + spent (§8 g). + + Without the flag it ENQUEUES one `code_rebuild` job and returns. It does not run the + rebuild here, and it emphatically does not stop the daemon: D7 requires the migration to + ride the single-writer supervisor as checkpointed background slices, and the job class it + enlarges is the one that wedged. `deploy` remains the separate owner-in-loop gate and is + untouched by this verb.""" + if dry_run: + import sqlite3 + + from core.ingest.code_corpus import build_code_corpus_sync + from ops.code_rebuild import measure_baseline, pending_commits + # `repo=self.repo_root`, not the default: `build_code_corpus_sync` otherwise resolves + # the repo from the CWD's git toplevel, so the measurement would silently describe + # whichever checkout the command was typed in rather than the one this run is pinned to. + sync = build_code_corpus_sync(self.cfg, repo=self.repo_root) + db_path = self.cfg.paths.data_dir / "code_snapshots.sqlite" + if not db_path.exists(): + print(f"code rebuild --dry-run: no ledger at {db_path} — nothing to measure.") + return 0 + db = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True) # read-only, said and meant + try: + base = measure_baseline(sync, db) + pending = len(pending_commits(db)) + finally: + db.close() + print(f"code rebuild --dry-run (READ-ONLY; nothing was written)\n {base}") + if base.embedder is not None: + print(f" embedder: live={base.embedder.live} stored_dim=" + f"{base.embedder.stored_dim} dim_match={base.embedder.dim_match} " + f"ledger_mismatched={base.embedder.ledger_mismatched} " + f"unrecorded_rows={base.embedder.unrecorded_rows}") + print(f" step 0: {pending} ledger commits still need commit_diffs capture") + return 0 + + from scheduler.code_sync import enqueue_code_rebuild + from scheduler.queue import JobQueue + from scheduler.router import Router + queue = JobQueue(self.cfg.paths.data_dir / "queue.sqlite") + try: + job = enqueue_code_rebuild(queue, Router(self.cfg)) + finally: + queue.close() + run = self.runs.last() + live = run is not None and run.active + where = ("the daemon will drain it as checkpointed BACKGROUND slices — `palace queue` to " + "watch." if live else + "no daemon is running — the job waits in the durable queue until `palace start`.") + print(f"code rebuild: enqueued code_rebuild job #{job.id}; {where}") + return 0 + # --- down / up / restart (KeepAlive-aware maintenance control, finding-0066) ------------- def _managed(self) -> bool: """Is the palace agent currently bootstrapped in its launchd domain (gui by default)?""" diff --git a/scheduler/code_sync.py b/scheduler/code_sync.py index 7851fd8..d4924a7 100644 --- a/scheduler/code_sync.py +++ b/scheduler/code_sync.py @@ -30,6 +30,7 @@ CODE_SYNC_KIND = "code_sync" CODE_BACKFILL_KIND = "code_backfill" # the history backfill (bp-099) — sibling of code_sync +CODE_REBUILD_KIND = "code_rebuild" # the D7 atom+membership rebuild (bp-153) — checkpointed Handler = Callable[[Job], "str | None"] @@ -70,6 +71,50 @@ def handle(_job: Job) -> str: return handle +def code_rebuild_handler(sync: CodeCorpusSync, db_path: Path, queue: JobQueue, *, + capture_budget_s: float = 60.0, + rebuild_budget_s: float = 120.0) -> Handler: + """The D7 rebuild (dn-vector-membership-store, bp-153) as a CHECKPOINTED background job. + + This is the first handler in the system to use the queue's `checkpoint`/resume protocol, and it + is the one the protocol was built for: the same lane wedged on 2026-07-25 because the whole + history rode one unbounded job (`code_sync` 300246, with 1,766 jobs queued behind it), and D7's + answer is slices with resume tokens and a per-slice time budget — never a monolith, and never a + daemon stop. + + One dispatch runs ONE slice. If work remains, the handler persists its resume token and + re-queues itself through `queue.checkpoint`, which also CLEARS the lease — so a yielded row + reads as waiting rather than as an orphan, and the next `claim` stamps a fresh per-batch + deadline (the shape §2.10 requires: "a healthy 14-hour backfill" must not die at hour N on a + per-job deadline). The supervisor completes the job only when it is still RUNNING after the + handler returns, so returning after a checkpoint yields rather than finishes. + + ⚑ It NEVER calls `CodeCorpusSync.backfill` — the old duplicated backfill is 52,755 embeds + against 22,502 atoms, 2.34× measured waste (D7), and this whole plan exists to not pay it.""" + def handle(job: Job) -> str | None: + from ops.code_rebuild import rebuild_step + from ops.code_snapshot import open_snapshot_db + db = open_snapshot_db(db_path) + try: + step = rebuild_step(sync, db, token=job.checkpoint, + capture_budget_s=capture_budget_s, + rebuild_budget_s=rebuild_budget_s) + finally: + db.close() + if not step.done: + queue.checkpoint(job.id, step.token or "") + return f"code rebuild [{step.phase}]: {step.message}" + return handle + + +def enqueue_code_rebuild(queue: JobQueue, router: Router) -> Job: + """Enqueue the D7 rebuild. Same pinned-tier, BACKGROUND species as its two siblings (it is a + model-less embed lane), and idempotent in the strong sense: every phase converges, so a + duplicate job re-derives at worst and re-lands nothing.""" + plan = router.plan(CODE_SYNC_KIND, priority=PRIORITY_BACKGROUND) + return queue.enqueue(CODE_REBUILD_KIND, plan.tier, plan.num_ctx, priority=plan.priority) + + def enqueue_code_backfill(queue: JobQueue, router: Router) -> Job: """Enqueue the history backfill. It reuses `code_sync`'s pinned-tier routing (the backfill is the same model-less species; `router._PINNED_KINDS` is out of this plan's write_scope, so we diff --git a/scripts/palace.py b/scripts/palace.py index 27f5c8a..6e083b6 100644 --- a/scripts/palace.py +++ b/scripts/palace.py @@ -13,6 +13,7 @@ uv run scripts/palace.py ingest-chat # on-demand: ingest the local Claude Code transcripts uv run scripts/palace.py code-seed # on-demand: seed the code embed lane (HEAD .py blobs) uv run scripts/palace.py code-backfill # on-demand: embed the full code history (bp-099) + uv run scripts/palace.py code-rebuild # the D7 atom+membership rebuild (bp-153); --dry-run uv run scripts/palace.py bless # owner-only: flip a plan proposed -> ready (gate) `start` seals the core (Invariant 1 — loopback only), runs preflight (ensures our own @@ -43,8 +44,8 @@ USAGE = ("usage: palace.py " "{start|stop|down|up|restart|status|queue|reset|deploy|ingest-chat|code-seed|" - "code-backfill|bless} " - "[--force] [--confirm] [--skip-tests] []") + "code-backfill|code-rebuild|bless} " + "[--force] [--confirm] [--skip-tests] [--dry-run] []") def bless(plan_id: str) -> int: @@ -240,6 +241,10 @@ def main(argv: list[str]) -> int: return launcher.code_seed() if cmd == "code-backfill": return launcher.code_backfill() + if cmd == "code-rebuild": + # ENQUEUES (or, with --dry-run, only measures). Never stops the daemon: D7 runs the + # migration as checkpointed queue slices under the single writer. + return launcher.code_rebuild(dry_run="--dry-run" in flags) print(USAGE) return 2 diff --git a/tests/integration/test_code_mirror.py b/tests/integration/test_code_mirror.py index 1493bb1..b1ea896 100644 --- a/tests/integration/test_code_mirror.py +++ b/tests/integration/test_code_mirror.py @@ -15,6 +15,7 @@ from __future__ import annotations import hashlib +from datetime import timedelta from typing import cast from core.ingest.code_corpus import code_memberships, code_rows, derive_code_chunks @@ -33,6 +34,7 @@ run_mc3, run_mc4, ) +from ops.code_rebuild import retire_legacy_rows from tests.fixtures.fakes import HashingEmbedder _DIM = 64 @@ -96,6 +98,47 @@ def test_ranked_paths_returns_only_code_paths(tmp_path): assert all(p == "core/store.py" for p, _ in ranked) +def test_the_firewall_survives_the_rebuilds_physical_maintenance(tmp_path): + """bp-153 Item 6: compaction and legacy-row retirement are PHYSICAL, and the mirror firewall + is a row prefilter — so the one thing that could quietly break it is a physical rewrite that + drops or rewrites the `provenance` column. + + The named degenerate input is a store where the default search returns nothing anyway: an empty + result satisfies "no code leaked" without testing anything. So the notes are asserted + RETRIEVABLE before and after, and the code lane is asserted still reachable through its own + explicit provenance set — a firewall that held by deleting the corpus is not a firewall.""" + store, memberships, emb = _mixed_store(tmp_path) + query = "nearest neighbour embedded chunks lancedb" + + before = semantic_search(query, cast(Embedder, emb), store, k=10) + assert before, "PRECONDITION: the mirror returns something, so 'no code' is not vacuous" + assert all(h["provenance"] != Provenance.CODE.value for h in before) + code_before = ranked_paths(query, emb, store, layers=LANE_LAYERS, pool=50, + memberships=memberships) + assert code_before, "PRECONDITION: the code lane is reachable before the rewrite" + rows_before = store.count() + versions_before = store.dataset_versions() + assert versions_before > 1, "PRECONDITION: there are dataset versions to compact away" + + # this store's code rows are already SHED atom rows, so there is no duplicated model to + # retire — asserted rather than assumed, so the compaction below is what is under test + assert retire_legacy_rows(store) == 0 + report = store.compact(older_than=timedelta(0)) + + assert report.rows_preserved, "compaction removed a logical row" + assert report.versions_dropped > 0, "...and it must actually have compacted something" + assert store.count() == rows_before, "no row was deleted" + + after = semantic_search(query, cast(Embedder, emb), store, k=10) + assert [h["id"] for h in after] == [h["id"] for h in before], "the mirror answer is unchanged" + assert all(h["provenance"] != Provenance.CODE.value for h in after) + assert ranked_paths(query, emb, store, layers=LANE_LAYERS, pool=50, + memberships=memberships) == code_before + + # the firewall is a PREFILTER, so the column it filters on must still be there to filter + assert all(str(r.get("provenance") or "") for r in store.project(["id", "provenance"])) + + def test_mc3_over_a_mixed_store_never_ranks_a_note(tmp_path): store, memberships, emb = _mixed_store(tmp_path) res = run_mc3(emb, store, probes=PROBES[:3], pool=50, memberships=memberships) diff --git a/tests/unit/test_code_ingest_wiring.py b/tests/unit/test_code_ingest_wiring.py index d38e396..6d79d6d 100644 --- a/tests/unit/test_code_ingest_wiring.py +++ b/tests/unit/test_code_ingest_wiring.py @@ -231,3 +231,187 @@ def test_palace_usage_lists_code_backfill() -> None: with redirect_stdout(buf): assert palace.main(["--help"]) == 0 assert "code-backfill" in buf.getvalue() + + +# --- bp-153 Item 5: the probe re-homes to membership fibers --------------------------------- + + +def _seed_ledger_two_versions(cfg) -> int: + """A ledger holding TWO distinct `(path, blob_sha)` versions. Two, not one, because the + counterfactual below turns on `1 < N`: with a single version the shed store's collapsed count + (1) equals the ledger's (1) and the old probe's false-positive is invisible.""" + from ops.code_snapshot import backfill as ledger_backfill + from ops.code_snapshot import open_snapshot_db + repo = cfg.paths.data_dir / "src" + repo.mkdir(parents=True, exist_ok=True) + _git(repo, "init", "-q", "-b", "main") + _git(repo, "config", "user.email", "t@t") + _git(repo, "config", "user.name", "t") + (repo / "m.py").write_text("def m():\n return 1\n") + (repo / "n.py").write_text("def n():\n return 2\n") + _git(repo, "add", "-A") + _git(repo, "commit", "-qm", "one") + db = open_snapshot_db(cfg.paths.data_dir / "code_snapshots.sqlite") + try: + ledger_backfill(db, repo) + finally: + db.close() + return 2 + + +def test_the_probe_does_not_false_positive_against_a_rebuilt_store(tmp_path) -> None: + """bp-153 Item 5 / the note's §6 re-home (1): the catch-up probe reads MEMBERSHIP FIBERS. + + This is finding-0166's named falsifier reappearing through a different door. bp-152 shed + `source_path` and `digest` from code atom rows, so the OLD probe — which counted distinct + `(source_path, digest)` pairs over the code lane — sees a rebuilt store collapse to the single + tuple `('', '')`. It then reads `1 < N` on every daemon start and enqueues a backfill FOREVER. + + Degenerate input: a store with no atom rows at all. The probe is then correct for the wrong + reason and this test would pass against the un-re-homed code. So the store here is REBUILT — + shed atom rows plus a complete set of fibers — and the OLD reading is computed alongside the + new one and asserted to be WRONG. Without that counterfactual the assertion is just "the probe + says complete", which the old code could also produce on some other input.""" + from core.ingest.code_corpus import code_memberships, code_rows, derive_code_chunks + from core.kernel.provenance import Provenance + from core.stores.memberships import EmbedderIdentity, open_membership_store + from core.stores.vectorstore import VectorStore + from ops.code_lineage import ledger_versions + from ops.code_snapshot import open_snapshot_db + from ops.lifecycle.launcher import _code_backfill_incomplete + from tests.fixtures.embedding import DIM, FakeEmbedder + + cfg = _cfg(tmp_path / "rebuilt", enabled=True) + n_versions = _seed_ledger_two_versions(cfg) + repo = cfg.paths.data_dir / "src" + + db = open_snapshot_db(cfg.paths.data_dir / "code_snapshots.sqlite") + try: + versions = ledger_versions(db) + finally: + db.close() + assert len(versions) == n_versions, "PRECONDITION: the ledger holds more than one version" + + # rebuild the store into the atom+membership model, exactly as bp-153's walk leaves it + vectors = VectorStore(cfg.paths.vector_store, dim=DIM) + memberships = open_membership_store(cfg) + embedder = FakeEmbedder() + for path, blob in versions: + source = _git(repo, "cat-file", "-p", blob) + chunks = derive_code_chunks(path, source) + vectors.add(code_rows(chunks, embedder.embed_documents([c.text for c in chunks]))) + memberships.write_fiber(code_memberships(path, blob, chunks)) + memberships.reconcile_currency(path, blob) + + # PRECONDITION: the store really is rebuilt — shed atom rows, and a fiber per ledger version + assert vectors.atom_row_count() > 0 + assert set(memberships.fibers()) == set(versions) + + # THE COUNTERFACTUAL: the old store-side reading collapses to ONE tuple and would loop forever + old_reading = {(str(r["source_path"]), str(r["digest"])) + for r in vectors.all_rows(provenances={Provenance.CODE})} + assert old_reading == {("", "")}, "the shed collapses every atom row to one bogus pair" + assert len(old_reading) < len(versions), "...so the OLD probe reads incomplete, forever" + + # the re-homed probe reads the fibers, and a complete store is exactly equal → no enqueue + from core.ingest.code_corpus import CodeCorpusSync + driver = CodeCorpusSync(repo=repo, store=vectors, embedder=embedder, + memberships=memberships, + embedder_identity=EmbedderIdentity(model="fake", dim=DIM)) + assert _code_backfill_incomplete(cfg, driver) is False + + # ...and it still says INCOMPLETE when a version is genuinely missing (it did not simply + # become a constant `False`, which would be the other way to stop the loop and be useless) + missing_cfg = _cfg(tmp_path / "missing", enabled=True) + _seed_ledger_two_versions(missing_cfg) + empty_driver = CodeCorpusSync( + repo=missing_cfg.paths.data_dir / "src", + store=VectorStore(missing_cfg.paths.vector_store, dim=DIM), embedder=embedder, + memberships=open_membership_store(missing_cfg), + embedder_identity=EmbedderIdentity(model="fake", dim=DIM)) + assert _code_backfill_incomplete(missing_cfg, empty_driver) is True + + +def test_the_backfill_triggers_are_unchanged_by_the_re_home(tmp_path) -> None: + """§6's fence: only the probe's DATA SOURCE moves. The cadence, the gate and the enqueued kind + stand — a trigger-level change would be out of design, and is the falsifier for Item 5.""" + inc = _cfg(tmp_path / "inc", enabled=True) + _seed_ledger(inc) + comps = build_components(inc) + try: # still enqueued by CATCH-UP, still exactly one, still the same kind + assert _catchup_kinds(comps).count(CODE_BACKFILL_KIND) == 1 + finally: + comps.queue.close() + + off_cfg = _cfg(tmp_path / "off", enabled=False) # still gated by ingestion.code.enabled + _seed_ledger(off_cfg) + off = build_components(off_cfg) + try: + assert _catchup_kinds(off).count(CODE_BACKFILL_KIND) == 0 + finally: + off.queue.close() + + +# --- bp-153 Item 7: palace code-rebuild -> a queued code_rebuild job ------------------------ + + +def test_code_rebuild_enqueues_one_job_and_registers_its_handler(tmp_path) -> None: + """Item 7: the verb ENQUEUES (single-writer: a job insert, never a store write from the CLI), + and the kind it enqueues has a handler registered — a job with no handler is a verb that + reaches nothing, which is the flag-off-is-not-done failure with extra steps.""" + from scheduler.code_sync import CODE_REBUILD_KIND + + cfg = _cfg(tmp_path, enabled=False) + launcher = Launcher(cfg=cfg, runs=RunLedger(tmp_path / "runs.sqlite"), + repo_root=Path(".").resolve()) + assert launcher.code_rebuild() == 0 + q = JobQueue(cfg.paths.data_dir / "queue.sqlite") + try: + assert [j.kind for j in q.list()].count(CODE_REBUILD_KIND) == 1 + finally: + q.close() + + comps = build_components(_cfg(tmp_path / "wired", enabled=False)) + try: + assert CODE_REBUILD_KIND in comps.supervisor.handlers # type: ignore[attr-defined] + finally: + comps.queue.close() + + +def test_code_rebuild_dry_run_writes_nothing_and_enqueues_nothing(tmp_path) -> None: + """`--dry-run` is Item 1's read-only pass. Degenerate input: a dry-run over an absent ledger + trivially writes nothing, so a REAL ledger is seeded first and the pass is asserted to have + measured it — while the queue stays empty and the stores stay untouched.""" + cfg = _cfg(tmp_path, enabled=True) + _seed_ledger_two_versions(cfg) + launcher = Launcher(cfg=cfg, runs=RunLedger(tmp_path / "runs.sqlite"), + repo_root=cfg.paths.data_dir / "src") + + buf = io.StringIO() + with redirect_stdout(buf): + assert launcher.code_rebuild(dry_run=True) == 0 + out = buf.getvalue() + + assert "READ-ONLY" in out and "nothing was written" in out + assert "versions=2" in out, "the pass actually measured the seeded ledger" + assert "ratio=" in out and "step 0:" in out + + q = JobQueue(cfg.paths.data_dir / "queue.sqlite") + try: + assert q.list() == [], "a dry run must enqueue NOTHING" + finally: + q.close() + + +def test_palace_usage_lists_code_rebuild() -> None: + """The ON switch must be reachable (finding-0159): the verb is in USAGE and in `--help`.""" + spec = importlib.util.spec_from_file_location( + "palace_cli", REPO_ROOT / "scripts" / "palace.py") + assert spec and spec.loader + palace = importlib.util.module_from_spec(spec) + spec.loader.exec_module(palace) + assert "code-rebuild" in palace.USAGE + buf = io.StringIO() + with redirect_stdout(buf): + assert palace.main(["--help"]) == 0 + assert "code-rebuild" in buf.getvalue() diff --git a/tests/unit/test_code_lineage.py b/tests/unit/test_code_lineage.py index 80ba84e..42af443 100644 --- a/tests/unit/test_code_lineage.py +++ b/tests/unit/test_code_lineage.py @@ -208,3 +208,80 @@ def test_composed_supersession_edge_resolves_to_embedded_nodes(ledger, repo, tmp path, old_blob, new_blob = modify assert sync.memberships.fiber(path, old_blob) assert sync.memberships.fiber(path, new_blob) + + +# ── bp-153 Item 5 — the `:139` docstring says what the code does ───────────────────────── + +@pytest.fixture +def revert_repo(tmp_path) -> Path: + """A repo whose single file goes v0 -> v1 -> **back to v0** — a real revert shape. + + Kept separate from the `repo` fixture above deliberately: that history is strictly increasing, + and on a strictly increasing history adjacent-collapse and distinct-collapse are the SAME + function. A revert is the only input that tells them apart, which is exactly why the docstring + could contradict the behavior for so long without any test noticing.""" + r = tmp_path / "revert_repo" + r.mkdir() + _git(r, "init", "-q", "-b", "main") + _git(r, "config", "user.email", "t@t") + _git(r, "config", "user.name", "t") + v0 = "def f():\n return 0\n" + v1 = "def f():\n return 1\n" + for text, msg in ((v0, "c1"), (v1, "c2 edit"), (v0, "c3 revert")): + (r / "f.py").write_text(text) + _git(r, "add", "-A") + _git(r, "commit", "-qm", msg) + return r + + +def test_a_revert_threads_as_three_runs_two_edges_not_two_distinct_blobs(revert_repo, tmp_path): + """§4 / F4 (file grain): `[A, B, A]` is PRESERVED — a revert stays visible. + + The named degenerate input is a history with no revert (the `repo` fixture above), where every + chain is strictly increasing and the claim is untestable. Here the chain is asserted to + genuinely revisit a blob first, and then both readings are computed side by side so the + assertion is a DIFFERENCE between them rather than a property one of them happens to have.""" + db = open_snapshot_db(tmp_path / "revert_snapshots.sqlite") + try: + ledger_backfill(db, revert_repo) + capture_commit_diffs(db, revert_repo, ledger_commits(db)) + chain = supersession_chains(db)["f.py"] + versions = {b for p, b in ledger_versions(db) if p == "f.py"} + finally: + db.close() + + # PRECONDITION: the history really reverts — two distinct blobs, one of them re-occupied + assert len(set(chain)) == 2, "the fixture must revisit a blob, or the two readings coincide" + assert chain[0] == chain[2] != chain[1] + + # the behavior: ADJACENT collapse keeps three runs and two edges + assert len(chain) == 3 + assert len(chain) - 1 == 2, "|edges| = |runs| - 1 (§4)" + + # the counterfactual, COMPUTED rather than described: a distinct collapse erases the revert + distinct = list(dict.fromkeys(chain)) + assert distinct == [chain[0], chain[1]] + assert len(distinct) - 1 == 1, "...one edge, and the revert is gone" + assert len(chain) != len(distinct), "the two readings differ on exactly this input" + + # the ledger is unaffected either way: a revert re-uses a blob, it does not mint a version + assert versions == set(chain) + assert len(versions) == 2 + + +def test_the_supersession_chains_docstring_does_not_contradict_its_own_code(): + """The `:139` correction, pinned as a ratchet (bp-153 Item 5). + + The docstring said a chain is "the ordered DISTINCT sequence of its blobs". The code collapses + only ADJACENT repeats, so the sentence described a different function — and the design note + cites this very behavior as evidence in its F4 dispute, which means the prose was contradicting + the argument that rests on it. This is the issue #28 defect class (a docstring drifting from + the thing it documents) caught where it was load-bearing, so it gets a test rather than a fix + and a hope. + + Degenerate input: asserting the word "adjacent" appears somewhere. The old text could have + gained that word and kept the wrong claim, so the CONTRADICTING phrase is asserted absent.""" + doc = supersession_chains.__doc__ or "" + assert "adjacent" in doc.lower(), "the docstring must name the collapse it performs" + assert "distinct sequence" not in doc.lower(), "...and must not claim the one it does not" + assert "[A, B, A]" in doc, "the revert case is the point, so it is spelled out" diff --git a/tests/unit/test_code_rebuild.py b/tests/unit/test_code_rebuild.py new file mode 100644 index 0000000..395909c --- /dev/null +++ b/tests/unit/test_code_rebuild.py @@ -0,0 +1,752 @@ +"""ops/code_rebuild.py — the D7 rebuild (bp-153): baseline, step-0 capture, the resumable walk, +compaction, and the owner-visible verb. + +Every criterion here names the **degenerate input** on which it would pass without testing its +claim, and asserts that precondition FIRST. The two that matter most, because both have a +false-success shape the design explicitly warns about: + + * **The dedup factor.** `|atoms| <= Sigma chunks` holds vacuously at zero savings (D7/§8 g), so + the fixture is asserted to contain atoms SHARED across versions and ACROSS FILES before any + ratio is read — and the measured factor, not the inequality, is what is asserted. + * **Compaction.** "row count and search results unchanged" is exactly what a compaction that + silently did nothing produces, so the version count is asserted to have DROPPED in the same + breath — and the fixture is asserted to hold several dataset versions before that can mean + anything. + +No Ollama: a deterministic counting fake embedder, temp stores, and a real git repo + real φ_code +ledger built per-test. Nothing here touches the live corpus. +""" + +from __future__ import annotations + +import json +import sqlite3 +import subprocess +from collections.abc import Iterator +from dataclasses import dataclass +from pathlib import Path + +import pytest + +from core.ingest.code_corpus import CodeCorpusSync, atom_id, derive_code_chunks +from core.stores.memberships import EmbedderIdentity, MembershipStore, frequency_gauges +from core.stores.vectorstore import LAYER_CODE_AST, LAYER_CODE_TEXT, VectorStore +from ops.code_lineage import ledger_versions, supersession_chains +from ops.code_rebuild import ( + PHASE_CAPTURE, + PHASE_COMPACT, + PHASE_LAND, + PHASE_SEED, + canonical_body_of_stored, + capture_slice, + carry_forward_candidates, + measure_baseline, + pending_commits, + rebuild_slice, + rebuild_step, + retire_legacy_rows, + seed_carry_forward, +) +from ops.code_snapshot import backfill as ledger_backfill +from ops.code_snapshot import open_snapshot_db +from tests.fixtures.embedding import DIM, FakeEmbedder + +_EMB = EmbedderIdentity(model="fake-embedder", dim=DIM) + +# A helper written IDENTICALLY into two files. This is the cross-file sharing PD-1 rules in, and it +# is what stops the dedup assertions below from being about one file's history alone. +_SHARED_HELPER = ( + "def helper(value):\n" + ' """Double it."""\n' + " return value * 2\n" +) + + +class _CountingEmbedder(FakeEmbedder): + """Counts embed calls. Reuse is INVISIBLE in the vectors — identical text gives an identical + vector whether it was recomputed or carried forward — so the call count is the only place the + carry-forward's "zero embeds" claim can actually be read.""" + + def __init__(self) -> None: + self.embedded: list[str] = [] + + def embed_documents(self, texts: list[str]) -> list[list[float]]: + self.embedded.extend(texts) + return super().embed_documents(texts) + + +def _git(repo: Path, *args: str) -> str: + return subprocess.run(["git", "-C", str(repo), *args], check=True, + capture_output=True, text=True).stdout + + +@pytest.fixture +def repo(tmp_path: Path) -> Path: + """A repo carrying every shape the rebuild's claims need: + + * `a.py` evolves v0 -> v1 -> **back to v0** (a real REVERT: the [A, B, A] chain shape); + * `b.py` holds a byte-identical copy of `a.py`'s helper (CROSS-FILE sharing, PD-1) and is + itself edited on main, so it has two versions; + * `c.py` is created AND THEN EDITED on a side branch before the merge — so its intermediate + blob is a ledger version that appears at NO commit on HEAD's first-parent line (D4/F3); + * every version re-lands the unchanged helper, so atoms are shared ACROSS versions too. + """ + r = tmp_path / "repo" + r.mkdir() + _git(r, "init", "-q", "-b", "main") + _git(r, "config", "user.email", "t@t") + _git(r, "config", "user.name", "t") + + def commit(msg: str) -> None: + _git(r, "add", "-A") + _git(r, "commit", "-qm", msg) + + v0 = _SHARED_HELPER + "\ndef work():\n return 1\n" + v1 = _SHARED_HELPER + "\ndef work():\n return 2\n" + (r / "a.py").write_text(v0) + (r / "b.py").write_text(_SHARED_HELPER + "\ndef other():\n return 9\n") + commit("c1") + + (r / "a.py").write_text(v1) # a.py -> B + commit("c2 edit work") + + (r / "a.py").write_text(v0) # a.py -> back to A: THE REVERT + commit("c3 revert work") + + _git(r, "checkout", "-q", "-b", "feat") + # TWO commits on the branch: the intermediate c.py blob never appears on HEAD's first-parent + # line, which is the side-branch shape D4/F3 is about. + (r / "c.py").write_text(_SHARED_HELPER + "\ndef side():\n return 3\n") + commit("feat add c") + (r / "c.py").write_text(_SHARED_HELPER + "\ndef side():\n return 4\n") + commit("feat edit c") + _git(r, "checkout", "-q", "main") + (r / "b.py").write_text(_SHARED_HELPER + "\ndef other():\n return 10\n") + commit("c4 edit b") + _git(r, "merge", "-q", "--no-ff", "feat", "-m", "merge feat") + return r + + +@pytest.fixture +def ledger(repo: Path, tmp_path: Path) -> Iterator[sqlite3.Connection]: + db = open_snapshot_db(tmp_path / "code_snapshots.sqlite") + ledger_backfill(db, repo) + yield db + db.close() + + +@dataclass +class Bench: + sync: CodeCorpusSync + embedder: _CountingEmbedder + + @property + def vectors(self) -> VectorStore: + return self.sync.store + + @property + def memberships(self) -> MembershipStore: + return self.sync.memberships + + +def _bench(repo: Path, root: Path, name: str = "b") -> Bench: + embedder = _CountingEmbedder() + return Bench( + sync=CodeCorpusSync( + repo=repo, store=VectorStore(root / f"{name}.lance", dim=DIM), embedder=embedder, + memberships=MembershipStore(root / f"{name}.sqlite"), embedder_identity=_EMB), + embedder=embedder) + + +@pytest.fixture +def bench(repo: Path, tmp_path: Path) -> Bench: + return _bench(repo, tmp_path) + + +# ── Item 1 — the read-only baseline ────────────────────────────────────────────────────── + + +def test_the_fixture_shares_atoms_across_versions_and_across_files( + bench: Bench, ledger: sqlite3.Connection) -> None: + """THE PRECONDITION for every dedup claim below, asserted once here rather than assumed. + + On a corpus where no atom is ever shared, `Sigma chunks == |atoms|`, the ratio is 1.0, and + every "the rebuild deduplicates" assertion passes while testing nothing. That is the + false-success shape D7 names, so the sharing is established as a FACT of this fixture first: + the same helper body appears in three different files and in several versions of one of them.""" + versions = ledger_versions(ledger) + assert len(versions) >= 6, "fixture must hold a real multi-version history" + + seen: dict[str, set[tuple[str, str]]] = {} + for path, blob in versions: + source = _git(bench.sync.repo, "cat-file", "-p", blob) + for ch in derive_code_chunks(path, source): + seen.setdefault(atom_id(ch), set()).add((path, blob)) + + # an atom shared ACROSS FILES (PD-1's cross-file dedup — the helper is byte-identical) + cross_file = [cid for cid, occ in seen.items() if len({p for p, _ in occ}) >= 3] + assert cross_file, "fixture must share an atom across at least three files" + + # an atom shared ACROSS VERSIONS of one file (the revert re-occupies its original atoms) + a_versions = {b for p, b in versions if p == "a.py"} + assert len(a_versions) == 2, "the revert means a.py has TWO distinct blobs, re-used three times" + multi_version = [cid for cid, occ in seen.items() + if len({b for p, b in occ if p == "a.py"}) >= 2] + assert multi_version, "fixture must share an atom across versions of one file" + + +def test_the_baseline_measures_the_factor_and_writes_nothing( + bench: Bench, ledger: sqlite3.Connection, tmp_path: Path) -> None: + """Item 1: the rebuild's target and cost, re-derived at the CURRENT cut, with no store written. + + Degenerate input: a pass that reports `|atoms| <= Sigma chunks` proves nothing — it holds at + zero savings. So the MEASURED FACTOR is asserted (strictly above 1, and equal to the ratio of + the two counts it reports), on a fixture whose sharing the test above established. + + "Writes nothing" is likewise asserted rather than trusted: the embedder would raise if called + (it is counted), and the store, the membership relation and the ledger file are all compared + before and after.""" + rows_before = bench.vectors.count() + m_before = bench.memberships.count() + ledger_path = tmp_path / "code_snapshots.sqlite" + ledger_mtime = ledger_path.stat().st_mtime_ns + ledger_size = ledger_path.stat().st_size + + base = measure_baseline(bench.sync, ledger) + + assert base.versions == len(ledger_versions(ledger)) + assert base.unreadable == 0 + assert base.chunks > 0 and base.atoms > 0 + assert base.atoms < base.chunks # necessary, and NOT sufficient + assert base.ratio == pytest.approx(base.chunks / base.atoms) + assert base.ratio > 1.0 # the measured factor is the claim + assert base.embeds_avoided == base.chunks - base.atoms + + # per lane, and the lanes are reported separately because they dedup differently + assert set(base.per_layer) >= {LAYER_CODE_AST, LAYER_CODE_TEXT} + assert sum(v.chunks for v in base.per_layer.values()) == base.chunks + assert sum(v.atoms for v in base.per_layer.values()) == base.atoms + + # READ-ONLY, asserted on all four surfaces + assert bench.embedder.embedded == [], "the baseline must never embed" + assert bench.vectors.count() == rows_before + assert bench.memberships.count() == m_before + assert ledger_path.stat().st_mtime_ns == ledger_mtime + assert ledger_path.stat().st_size == ledger_size + + +def test_the_baseline_reports_the_embedder_identity_it_can_and_cannot_see( + bench: Bench, ledger: sqlite3.Connection) -> None: + """Item 1's last column: is the carry-forward seed's geometry the live one? + + The degenerate answer is "yes" returned unconditionally. So the check is asserted to be + STRUCTURED: `dim` is recoverable and is compared; `model` is NOT recorded on a pre-D1 row (the + Arrow schema is shared with the prose lane and has no embedder column), and the count of rows + carrying that status is reported rather than silently treated as a match.""" + bench.sync.sync() # land HEAD so the plane is non-empty + base = measure_baseline(bench.sync, ledger) + check = base.embedder + assert check is not None + assert check.live == _EMB + assert check.stored_dim == DIM # recoverable, and compared + assert check.dim_match is True + assert check.ledger_mismatched == 0 + assert check.matches is True + + # the falsifier: a dimension that contradicts the live config is NOT a match + from ops.code_rebuild import EmbedderCheck + mismatched = EmbedderCheck(live=_EMB, stored_dim=DIM + 1) + assert mismatched.dim_match is False and mismatched.matches is False + + +# ── Item 2 — step 0: the sliced `commit_diffs` capture ─────────────────────────────────── + + +def test_the_sliced_capture_is_idempotent_and_leaves_durable_progress( + bench: Bench, ledger: sqlite3.Connection, repo: Path) -> None: + """Item 2: the first successful capture, and R6's falsifier as a passing property. + + R6 names the failure as "a slice exceeding its budget WITHOUT leaving a checkpoint". The + degenerate input is a budget so generous the slice never yields — that tests nothing about + slicing — so the budget here is ZERO, forcing a yield at the first boundary, and the durable + progress is then read back out of the marker table rather than out of a return value.""" + total = len(pending_commits(ledger)) + assert total >= 5, "fixture must hold enough commits for a slice to yield mid-history" + + first = capture_slice(ledger, repo, budget_s=0.0, batch=1) + assert first.budget_spent is True and first.done is False + assert first.captured >= 1, "a yielding slice must still make progress, never livelock" + + # THE R6 PROPERTY: the progress is durable, on disk, before the yield — not held in the token + marked = ledger.execute("SELECT count(*) FROM _commit_diffs_captured").fetchone()[0] + assert marked == first.captured + assert len(pending_commits(ledger)) == total - first.captured + + while True: # resume to completion, one slice at a time + step = capture_slice(ledger, repo, budget_s=0.0, batch=2) + if step.done: + break + assert pending_commits(ledger) == [] + assert ledger.execute("SELECT count(*) FROM _commit_diffs_captured").fetchone()[0] == total + assert ledger.execute("SELECT count(*) FROM commit_diffs").fetchone()[0] > 0 + + # idempotence: a re-run captures ZERO (the marker table, `:119-129`) + again = capture_slice(ledger, repo, budget_s=30.0) + assert again.captured == 0 and again.done is True + + +def test_pending_commits_reads_a_read_only_ledger_without_creating_its_schema( + ledger: sqlite3.Connection, repo: Path, tmp_path: Path) -> None: + """The dry-run's read must not write — found by the acceptance test, so it gets a ratchet. + + Degenerate input: asserting `pending_commits` works on a read-WRITE connection. It does, and it + used to do so by calling `_ensure_schema` first — a `CREATE TABLE` that silently mutated the + file a dry run promised to leave alone. On the LIVE ledger those tables have never existed, so + this is the ordinary path, not a corner. The test therefore opens the ledger `mode=ro`, where + the write cannot hide: it raises.""" + path = tmp_path / "code_snapshots.sqlite" + # PRECONDITION: the marker table genuinely does NOT exist yet — the live ledger's own state + assert ledger.execute( + "SELECT 1 FROM sqlite_master WHERE name = '_commit_diffs_captured'").fetchone() is None + + ro = sqlite3.connect(f"file:{path}?mode=ro", uri=True) + try: + pending = pending_commits(ro) # must not raise: no schema is created + assert pending, "an uncaptured ledger reports every commit pending" + assert len(pending) == len(ro.execute("SELECT commit_sha FROM snapshots").fetchall()) + finally: + ro.close() + + # ...and the file really was not written: the tables still do not exist + assert ledger.execute( + "SELECT 1 FROM sqlite_master WHERE name = '_commit_diffs_captured'").fetchone() is None + + # after a real capture, the same read reports the remainder — it is not simply "everything" + capture_slice(ledger, repo, budget_s=30.0) + assert pending_commits(ledger) == [] + + +def test_the_captured_chains_preserve_a_revert_as_three_runs_two_edges( + ledger: sqlite3.Connection, repo: Path) -> None: + """Item 2 + Item 5: `[A, B, A]` survives, so the `:139` docstring's corrected wording is the + behavior and not a wish. + + Degenerate input: a history with no revert. Every chain is then strictly increasing, and + adjacent-collapse is indistinguishable from distinct-collapse — the exact confusion the old + docstring licensed and the F4 dispute turns on. So the REVERT is asserted to exist in the + chain first (a.py returns to a blob it already left), and only then is the shape read.""" + capture_slice(ledger, repo, budget_s=30.0) + chains = supersession_chains(ledger) + chain = chains["a.py"] + + # PRECONDITION: the chain genuinely REVISITS a blob, non-adjacently + assert len(chain) == 3, "a.py: v0 -> v1 -> v0 is three RUNS" + assert chain[0] == chain[2] != chain[1], "...and the third run re-occupies the first blob" + assert len(set(chain)) == 2, "...over only TWO distinct blobs — which is what a revert IS" + + # the claim: adjacent-collapse keeps the revert; distinct-collapse would erase it + assert len(chain) - 1 == 2, "|edges| = |runs| - 1" + assert len(dict.fromkeys(chain)) == 2, "a DISTINCT collapse would report 2 runs / 1 edge" + assert len(chain) != len(dict.fromkeys(chain)), "...so the two readings genuinely differ here" + + +# ── Item 3 — the carry-forward seed and the sliced, resumable walk ─────────────────────── + + +def test_the_carry_forward_seed_lands_atoms_at_zero_embeds(bench: Bench, + ledger: sqlite3.Connection) -> None: + """Item 3's economics: ~2/3 of the atom set enters by canonical RE-HASH of rows the old model + already embedded, at zero embedder calls. + + Degenerate input: a store with no pre-D1 rows seeds nothing and "costs zero embeds" trivially. + So a legacy-shaped store is built first and the recoverable set is asserted NON-EMPTY before + the zero-embed claim is read.""" + legacy = _legacy_store(bench, ledger) + targets = {atom_id(ch) for path, blob in ledger_versions(ledger) + for ch in derive_code_chunks(path, _git(bench.sync.repo, "cat-file", "-p", blob))} + + seed_ids = carry_forward_candidates(legacy, targets=targets) + # PRECONDITION: there is something to carry forward, and it is a real fraction of the target set + assert seed_ids, "the legacy store must hold rows whose canonical body re-hashes to a target" + assert len(seed_ids) < len(targets), "...but not all of them — L1 does not survive (D7)" + assert {cid.split(":", 1)[0] for cid in seed_ids} == {LAYER_CODE_AST, LAYER_CODE_TEXT} + + # The seed reads and writes ONE store — which is the production shape, not a shortcut: D1 + # evolves the existing `chunks` table in place, so the legacy rows and the atom rows the seed + # mints from them live side by side in the same table. + memberships = MembershipStore(legacy.path.parent / "seeded.sqlite") + atoms_before = legacy.atom_row_count() + rows_before = legacy.count() + assert atoms_before == 0, "PRECONDITION: no atom rows yet — the seed is what creates them" + + report = seed_carry_forward(legacy, memberships, _EMB, targets=seed_ids, stored_dim=DIM) + assert report.refused_embedder_mismatch is False + assert report.rows_written == len(seed_ids) + assert report.atoms_recorded == len(seed_ids) + assert report.embeds == 0 + + # the seeded atoms are now PRESENT under the live identity, so a land reuses them + assert memberships.known_atoms(seed_ids, _EMB) == seed_ids + assert legacy.atom_row_count() == len(seed_ids) + assert legacy.count() == rows_before + len(seed_ids), "APPEND-only: no legacy row was replaced" + + +def test_the_seed_refuses_outright_when_the_embedder_identity_disagrees( + bench: Bench, ledger: sqlite3.Connection) -> None: + """The embedder pin as a REFUSAL, not a warning (bp-152 §6, owner-confirmed). + + Degenerate input: a suite that only ever exercises one embedder cannot see this at all — the + seed would happily copy vectors from another geometry into the one ANN space, and no downstream + measurement could detect it. So a contradicting dimension is injected and the seed must do + NOTHING: not fewer rows, none.""" + legacy = _legacy_store(bench, ledger) + targets = {atom_id(ch) for path, blob in ledger_versions(ledger) + for ch in derive_code_chunks(path, _git(bench.sync.repo, "cat-file", "-p", blob))} + seed_ids = carry_forward_candidates(legacy, targets=targets) + assert seed_ids, "PRECONDITION: there is a seed to refuse" + + memberships = MembershipStore(legacy.path.parent / "refused.sqlite") + report = seed_carry_forward(legacy, memberships, _EMB, targets=seed_ids, stored_dim=DIM + 1) + assert report.refused_embedder_mismatch is True + assert report.rows_written == 0 and report.atoms_recorded == 0 + assert legacy.atom_row_count() == 0, "a PARTIAL seed would be the corruption" + assert memberships.ledger_atom_ids() == set(), "...and nothing was recorded as reusable" + + +def test_killing_the_rebuild_mid_slice_and_resuming_lands_exactly_once( + bench: Bench, ledger: sqlite3.Connection, repo: Path, tmp_path: Path) -> None: + """Item 3's central property: sliced, checkpointed, RESUMABLE, with no double-landing. + + Degenerate input: a "resume" that re-runs from scratch into an empty store also produces a + correct final state — it would prove nothing about resumption. So the run is genuinely + interrupted (a zero budget forces a yield with real work already landed, asserted), resumed + from its token, and the result compared against an INDEPENDENT single-shot rebuild of the same + ledger: identical |V|, identical |M|, identical fibers, chunk for chunk. + + Fiber equality is why this holds — derivation is pure, so a re-landed version writes no row — + which is also why the assertion is on the fibers themselves and not merely on the counts.""" + sliced = bench + progress = rebuild_slice(sliced.sync, ledger, token=None, budget_s=0.0, slice_size=1) + # PRECONDITION: the run really was interrupted with work already done + assert progress.token is not None, "a zero budget must yield a resume token" + assert progress.versions_landed >= 1, "...after landing something, so this is a real resume" + assert progress.remaining > 0 + mid_m = sliced.memberships.count() + assert mid_m > 0 + + token: str | None = progress.token + guard = 0 + while token is not None: + guard += 1 + assert guard < 200, "the walk must terminate" + step = rebuild_slice(sliced.sync, ledger, token=token, budget_s=0.0, slice_size=1) + token = step.token + assert sliced.memberships.count() > mid_m # it genuinely continued + + # the independent control: one uninterrupted rebuild of the same ledger + whole = _bench(repo, tmp_path, "whole") + done = rebuild_slice(whole.sync, ledger, token=None, budget_s=3600.0) + assert done.done is True + + assert sliced.memberships.count() == whole.memberships.count() + assert sliced.vectors.atom_row_count() == whole.vectors.atom_row_count() + assert sliced.memberships.fibers() == whole.memberships.fibers() + for path, blob in whole.memberships.fibers(): + assert sliced.memberships.fiber(path, blob) == whole.memberships.fiber(path, blob) + + +def test_a_resumed_rebuild_re_lands_idempotently_even_with_a_lost_token( + bench: Bench, ledger: sqlite3.Connection) -> None: + """The token is an optimization, never the safety property (R6/D2 step 3). + + Degenerate input: asserting resumability only along the happy path, where the token always + survives. The real crash loses it — so here the walk is re-run from scratch over an + already-complete store, and nothing may move: no embed, no new occupancy, no new atom.""" + rebuild_slice(bench.sync, ledger, budget_s=3600.0) + m_after, v_after = bench.memberships.count(), bench.vectors.atom_row_count() + assert m_after > 0 and v_after > 0 # PRECONDITION: something was landed + embeds = len(bench.embedder.embedded) + + replay = rebuild_slice(bench.sync, ledger, token=None, budget_s=3600.0) + assert replay.membership_rows == 0, "a re-land writes NO new occupancy" + assert replay.atoms_embedded == 0, "...and embeds nothing" + assert bench.memberships.count() == m_after + assert bench.vectors.atom_row_count() == v_after + assert len(bench.embedder.embedded) == embeds + + # an UNREADABLE token restarts the walk rather than failing it — same idempotent outcome + garbage = rebuild_slice(bench.sync, ledger, token="{not json", budget_s=3600.0) + assert garbage.membership_rows == 0 and garbage.atoms_embedded == 0 + assert bench.memberships.count() == m_after + + +def test_the_rebuilt_store_reproduces_the_baselines_measured_factor( + bench: Bench, ledger: sqlite3.Connection) -> None: + """§8(g): the rebuild lands the atom count Item 1 derived, against the duplicated model's count. + + Degenerate input, stated by the design itself: `|atoms| <= Sigma chunks` holds at zero savings. + So the assertion is an EQUALITY on the factor — the standing `|M|/|V|` gauge must reproduce the + baseline's ratio exactly, because `|M|` IS Sigma per-version chunks (one occupancy per chunk per + version) and `|V|` IS the distinct atom count. That is the same number arrived at from two + independent directions: derived-and-counted before the run, stored-and-queried after it.""" + base = measure_baseline(bench.sync, ledger) + assert base.ratio > 1.0, "PRECONDITION: the fixture actually dedups (see the sharing test)" + + done = rebuild_slice(bench.sync, ledger, budget_s=3600.0) + assert done.done is True + + gauges = frequency_gauges(bench.vectors, bench.memberships) + assert gauges.occupancies == base.chunks, "|M| == Sigma per-version chunks" + assert gauges.plane_atoms == base.atoms, "|V| == the distinct atom count" + assert gauges.dedup_factor == pytest.approx(base.ratio) + assert gauges.embeds_avoided == base.embeds_avoided + + # the embedder was called once per atom and never twice — the reuse is real, not incidental + assert len(bench.embedder.embedded) == base.atoms + assert len(set(bench.embedder.embedded)) == len(bench.embedder.embedded) + + # every ledger version has a fiber, INCLUDING the side-branch one that sits on no chain (F3) + assert len(bench.memberships.fibers()) == len(ledger_versions(ledger)) + + +def test_the_rebuild_lands_a_fiber_for_every_ledger_version_including_side_branch_ones( + bench: Bench, ledger: sqlite3.Connection, repo: Path) -> None: + """D4/F3: the version set is what the rebuild must cover — NOT the versions reachable along + HEAD's first-parent line. A walk that followed the merge history would silently drop the + intermediate blob a side branch produced, and that blob is a real member of M. + + Degenerate input: a linear history, where the two sets coincide and the distinction cannot + bite. The fixture's two-commit side branch is asserted to have produced exactly that gap first. + + ⚑ Note what this does NOT assert. The design's F3 says "chain members are a strict subset of + the version set", meaning `supersession_chains` misses side-branch versions. Measured against + the shipped reader that is FALSE — chains are threaded over every SNAPSHOTTED commit, each + diffed against its own first parent, so a side branch's own linear history is captured too + (verified on the live ledger: 1,663 versions, 1,663 chain members, an empty difference both + ways — issue filed). The claim that survives measurement, and the one the rebuild actually + depends on, is this one: HEAD's first-parent line is a strict subset of the ledger.""" + capture_slice(ledger, repo, budget_s=30.0) + versions = set(ledger_versions(ledger)) + + # the versions visible along HEAD's FIRST-PARENT line alone (what a merge-history walk sees) + first_parent: set[tuple[str, str]] = set() + for sha in _git(repo, "rev-list", "--first-parent", "HEAD").split(): + for line in _git(repo, "ls-tree", "-r", sha).splitlines(): + meta, _, path = line.partition("\t") + if path.endswith(".py"): + first_parent.add((path, meta.split()[2])) + + # PRECONDITION: the side branch really did produce a version off that line + off_line = versions - first_parent + assert off_line, "fixture must hold a version absent from HEAD's first-parent line" + assert first_parent < versions, "...so the first-parent view is a STRICT subset" + + rebuild_slice(bench.sync, ledger, budget_s=3600.0) + landed = set(bench.memberships.fibers()) + assert landed == versions, "every ledger version landed a fiber" + assert off_line <= landed, "...including the side-branch one" + + +# ── Item 6 — compaction and old-version cleanup (§3) ───────────────────────────────────── + + +def test_compaction_drops_versions_while_rows_and_search_are_unchanged( + bench: Bench, ledger: sqlite3.Connection) -> None: + """Item 6/§3: compaction is PHYSICAL — it removes no logical row and changes no answer. + + Degenerate input, named in the plan: a no-op compaction "changes nothing" and passes any test + asserting only that rows and search results are unchanged. So the version count is asserted to + have genuinely DROPPED, and the fixture is asserted to hold several dataset versions before + that drop can mean anything. Both halves, or neither is evidence.""" + rebuild_slice(bench.sync, ledger, budget_s=3600.0) + + # PRECONDITION: many write batches happened, so there is version accumulation to compact + before_versions = bench.vectors.dataset_versions() + assert before_versions > 2, "fixture must accumulate dataset versions for a drop to be visible" + + query = FakeEmbedder().embed_query("def helper(value):") + rows_before = bench.vectors.count() + hits_before = [str(h["id"]) for h in bench.vectors.search(query, k=5)] + atoms_before = bench.vectors.atom_row_count() + assert hits_before, "PRECONDITION: search returns something to be preserved" + + report = bench.vectors.compact(older_than=__import__("datetime").timedelta(0)) + + assert report.versions_dropped > 0, "the version count must actually FALL" + assert report.versions_after < report.versions_before + assert bench.vectors.dataset_versions() < before_versions + + assert report.rows_preserved is True # no logical row disappeared + assert report.rows_before == report.rows_after == rows_before + assert bench.vectors.count() == rows_before + assert bench.vectors.atom_row_count() == atoms_before + assert [str(h["id"]) for h in bench.vectors.search(query, k=5)] == hits_before + + # ...and the membership relation, which compaction never touches, is intact + assert frequency_gauges(bench.vectors, bench.memberships).orphans == 0 + + +def test_retiring_the_legacy_rows_supersedes_them_and_deletes_nothing( + bench: Bench, ledger: sqlite3.Connection) -> None: + """The rebuild lands the atom plane into the table holding the rows it replaces, so the old + per-version rows must stop answering the default current-view search — by SUPERSESSION + (keep-and-link, D2), never by deletion. + + Degenerate input: an empty legacy set makes "nothing was deleted" vacuous, so a populated + legacy store is asserted first, and the retained-row count is asserted after.""" + legacy = _legacy_store(bench, ledger) + total_before = legacy.count() + legacy_current = len(legacy.project(["id"], where="source_path <> '' AND current = true")) + assert legacy_current > 0, "PRECONDITION: there are current legacy rows to retire" + + flipped = retire_legacy_rows(legacy) + assert flipped == legacy_current + assert legacy.count() == total_before, "RETAINED — nothing was deleted (|V| cannot fall)" + assert len(legacy.project(["id"], where="source_path <> '' AND current = true")) == 0 + + assert retire_legacy_rows(legacy) == 0 # idempotent: a second call is a no-op + + +# ── Item 7 — `palace code-rebuild`, the owner-visible verb ─────────────────────────────── + + +def test_the_code_rebuild_verb_is_listed_and_dispatched() -> None: + """Item 7: wiring IS the deliverable — the ON path must exist, not merely the code behind it. + + Degenerate input: asserting the `Launcher` method exists. That is the "flag-off is not done" + failure exactly — a method nothing routes to is not a verb. So the USAGE line, the module + docstring's verb list, and the `main()` dispatch are all read, and the dispatch is read from + the SOURCE so a method that is never reached cannot pass.""" + import scripts.palace as palace + + assert "code-rebuild" in palace.USAGE + assert "code-rebuild" in (palace.__doc__ or "") + source = Path(palace.__file__).read_text(encoding="utf-8") + assert 'if cmd == "code-rebuild":' in source + assert "launcher.code_rebuild(dry_run=" in source + + from ops.lifecycle.launcher import Launcher + assert callable(Launcher.code_rebuild) + + # the kind is registered as a handler on the daemon, or the enqueued job would never dispatch + from scheduler.code_sync import CODE_REBUILD_KIND, code_rebuild_handler, enqueue_code_rebuild + assert CODE_REBUILD_KIND == "code_rebuild" + assert callable(code_rebuild_handler) and callable(enqueue_code_rebuild) + launcher_src = Path(__file__).resolve().parents[2] / "ops" / "lifecycle" / "launcher.py" + wiring = launcher_src.read_text(encoding="utf-8") + assert "CODE_REBUILD_KIND: code_rebuild_handler(" in wiring + + +def test_the_verb_never_stops_the_daemon() -> None: + """The invariant the design states outright: the rebuild is a queue citizen. A verb that + stopped the daemon would satisfy every functional assertion above and violate D7's central + operational claim, so the method's source is read for the calls it must not make.""" + import inspect + + from ops.lifecycle.launcher import Launcher + source = inspect.getsource(Launcher.code_rebuild) + for forbidden in ("self.stop(", "self.down(", "self.up(", "self.restart(", "self.deploy("): + assert forbidden not in source, f"code-rebuild must never call {forbidden}" + assert "enqueue_code_rebuild" in source, "...it ENQUEUES" + + +# ── the phase machine: one slice per dispatch, and the token that positions the next ───── + + +def test_the_phase_machine_walks_capture_then_seed_then_land_then_compact( + bench: Bench, ledger: sqlite3.Connection) -> None: + """The queue slice's shape (D7/S3): each dispatch runs ONE phase-slice and hands back the token + that positions the next, so the rebuild never rides a single unbounded job — which is exactly + how this lane wedged before. + + Degenerate input: a machine that reports the phases without doing them. So the OBSERVABLE + effect of each phase is asserted as it passes: capture populates `commit_diffs`, seed/land fill + the plane and the relation, compact reduces the dataset version count.""" + seen: list[str] = [] + token: str | None = None + guard = 0 + while True: + guard += 1 + assert guard < 300, "the phase machine must terminate" + step = rebuild_step(bench.sync, ledger, token=token, + capture_budget_s=0.0, rebuild_budget_s=0.0, + capture_batch=1, slice_size=1) + seen.append(step.phase) + if step.done: + break + token = step.token + assert token is not None + assert json.loads(token)["phase"] in {PHASE_CAPTURE, PHASE_SEED, PHASE_LAND, PHASE_COMPACT} + + # every phase ran, in the D7/§3 order, and the final one ended the walk + assert seen[0] == PHASE_CAPTURE + assert seen[-1] == PHASE_COMPACT + order = [p for i, p in enumerate(seen) if i == 0 or seen[i - 1] != p] + assert order == [PHASE_CAPTURE, PHASE_SEED, PHASE_LAND, PHASE_COMPACT] + # a zero budget with batch/slice 1 must have forced BOTH long phases to yield repeatedly — + # otherwise the walk fitted in one dispatch and nothing about slicing has been exercised + assert seen.count(PHASE_CAPTURE) > 1, "capture must yield mid-phase and resume" + assert seen.count(PHASE_LAND) > 1, "the land walk must yield mid-phase and resume" + + # the observable effects, phase by phase + assert pending_commits(ledger) == [] + assert ledger.execute("SELECT count(*) FROM commit_diffs").fetchone()[0] > 0 + base = measure_baseline(bench.sync, ledger) + gauges = frequency_gauges(bench.vectors, bench.memberships) + assert gauges.occupancies == base.chunks + assert gauges.plane_atoms == base.atoms + assert gauges.dedup_factor == pytest.approx(base.ratio) + + +# ── the canonical-body recovery the seed depends on ────────────────────────────────────── + + +def test_canonical_recovery_refuses_a_body_line_that_merely_looks_like_a_header() -> None: + """A wrong strip is silent identity corruption — it mints an atom nothing can ever hit again. + + Degenerate input: a body whose first line does not begin with `#` at all. Any implementation + handles that. The dangerous case is a body whose first line IS a comment, which is ordinary in + this repo (`ops/code_lineage.py` opens with one), so the recovery is verified against the row's + OWN `source_path` rather than against the shape of the line.""" + body = "def f():\n return 1\n" + header = "# pkg/mod.py:f(x)" + assert canonical_body_of_stored(LAYER_CODE_AST, f"{header}\n{body}", "pkg/mod.py") == body + + # a real first-line comment, under a row whose path it does NOT name: refused, not stripped + commented = "# ── Family 1 boundary ──\ndef f():\n return 1\n" + assert canonical_body_of_stored(LAYER_CODE_AST, commented, "pkg/mod.py") is None + + # L0b windows are headerless by construction, so the window IS its canonical body + assert canonical_body_of_stored(LAYER_CODE_TEXT, commented, "pkg/mod.py") == commented + + # L1 never carries forward: its windows were cut over header-bearing prose (D7) + assert canonical_body_of_stored("codedoc", f"# pkg/mod.py\n{body}", "pkg/mod.py") is None + + +# ── helpers ────────────────────────────────────────────────────────────────────────────── + + +def _legacy_store(bench: Bench, ledger: sqlite3.Connection) -> VectorStore: + """A store shaped like the LIVE one: pre-D1 rows carrying `source_path`/`digest` and a + header-bearing embed text, one per chunk per version — the duplicated model this replaces. + + Built by hand rather than by calling the old lander, because the old lander no longer exists: + the point is to reproduce the ROWS the live store actually holds, so the carry-forward's + canonical re-hash is exercised against real shapes.""" + from core.kernel.provenance import Provenance + + store = VectorStore(bench.vectors.path.parent / "legacy.lance", dim=DIM) + embedder = FakeEmbedder() + rows: list[dict[str, object]] = [] + for path, blob in ledger_versions(ledger): + source = _git(bench.sync.repo, "cat-file", "-p", blob) + for i, ch in enumerate(derive_code_chunks(path, source)): + rows.append({ + "id": f"{path}:{ch.content_hash}", "digest": blob, "title": path, + "source_path": path, "chunk_index": i, "provenance": Provenance.CODE.value, + "text": ch.text, "layer": ch.layer, "qualname": ch.qualname, + "line_start": ch.slot_line_start, "line_end": ch.slot_line_end, + "current": True, "vector": embedder.embed_documents([ch.text])[0], + }) + store.add(rows) + return store diff --git a/tests/unit/test_memberships.py b/tests/unit/test_memberships.py index 7ce4dcc..51c1cd5 100644 --- a/tests/unit/test_memberships.py +++ b/tests/unit/test_memberships.py @@ -45,6 +45,7 @@ Membership, MembershipStore, current_any_drift, + frequency_gauges, purge_atom, repair_current_any, resolve_occupancies, @@ -685,3 +686,127 @@ def test_the_vector_plane_never_shrinks_except_across_a_purge(degenerate: Fixtur report = purge_atom(b.vectors, b.memberships, degenerate.shared_id) assert report.vector_rows_deleted == 1 # the ONE exception, and it is logged assert b.n_vectors() == before - 1 + + +# ── bp-153 Item 4 — the frequency-plane gauges (D6) ────────────────────────────────────── + +def test_n_doc_and_n_occ_differ_on_the_duplicate_window_pair(degenerate: Fixture) -> None: + """D6/F5: the two counts are different QUESTIONS and must never be conflated. + + The named degenerate input is a corpus with no repeated window — there `n_doc == n_occ` for + every atom, and a conflated implementation passes every assertion. So the DUPLICATE L0b PAIR + is asserted to exist first: `dup` occupies blob A's fiber twice with distinct `chunk_index` + (the multiset pin), in ONE path. That is the only shape on which the two readings can be told + apart, and on it they must disagree: `n_occ` counts both rows, `n_doc` counts one path.""" + m = degenerate.bench.memberships + dup = degenerate.dup_id + + # PRECONDITION — without a duplicate pair in a single fiber this test cannot see the defect. + pair = [x for x in m.fiber("pkg/w.py", "A") if x.content_id == dup] + assert len(pair) == 2, "fixture must hold a duplicate L0b window pair" + assert {x.chunk_index for x in pair} == {2, 3}, "...as TWO rows, distinct by chunk_index" + assert len({x.path for x in pair}) == 1, "...inside ONE path, so n_doc cannot see them both" + + # the claim: the multiset reading counts both occupancies, the document reading counts one path + assert m.n_occ(dup) == 2 + assert m.n_doc(dup) == 1 + assert m.n_occ(dup) != m.n_doc(dup) + + # lifetime, where every fiber that ever held it counts: A, B and the side branch S + assert m.n_occ(dup, current_only=False) == 6 + assert m.n_doc(dup, current_only=False) == 1 + + # the control that makes the inequality mean something: a NON-repeated atom agrees on both + # readings in one path, so the difference above is the repetition and not an off-by-one + assert m.n_occ(degenerate.shared_id) == m.n_doc(degenerate.shared_id) + + +def test_the_dedup_factor_is_the_d7_falsifier_kept_observable(degenerate: Fixture) -> None: + """D6/S5: `|M|/|V|` IS the dedup factor, so D7's economics stay checkable forever. + + It fails its keep by sitting at ≈1.0 after a rebuild — the model bought nothing. A test that + only asserted `dedup > 1` on a fixture that happens to share atoms would not establish that, + so the CONTROL is built explicitly: a second store where every landing is a fresh atom reads + ≈1.0, and the gauge separates the two.""" + b = degenerate.bench + g = frequency_gauges(b.vectors, b.memberships) + + # PRECONDITION: real reuse exists — occupancies outnumber atoms, and a shared atom spans paths + assert g.occupancies > g.atoms, "fixture must actually reuse atoms" + assert b.memberships.n_doc(degenerate.shared_id, current_only=False) == 2 + + assert g.occupancies == b.memberships.count() # |M| is the relation, whole + assert g.plane_atoms == b.n_vectors() # |V| is the plane, whole + assert g.dedup_factor == pytest.approx(g.occupancies / g.plane_atoms) + assert g.dedup_factor > 1.0 + assert g.embeds_avoided == g.occupancies - g.atoms + assert g.orphans == 0 # every atom here has an occupancy + + # per lane, and the lanes genuinely differ: L0b holds the duplicate pair, so its factor is the + # higher one — a gauge reporting one number for both lanes could not show that. + assert set(g.per_layer) == {LAYER_CODE_AST, LAYER_CODE_TEXT} + assert g.per_layer[LAYER_CODE_TEXT].dedup_factor > g.per_layer[LAYER_CODE_AST].dedup_factor + for lane in g.per_layer.values(): + assert lane.embeds_avoided == lane.occupancies - lane.atoms + + # THE CONTROL: a store that never reuses anything reads ≈1.0 — this is what "the model bought + # nothing" looks like, and it is the reading the falsifier names. + fresh_v = VectorStore(b.vectors.path.parent / "control.lance", dim=DIM) + fresh_m = MembershipStore(b.memberships.path.parent / "control.sqlite") + control = CodeLander(vectors=fresh_v, memberships=fresh_m, + embedder=_CountingEmbedder(), embedder_identity=_EMB) + for i, blob in enumerate(("c1", "c2", "c3")): + control.land("only.py", blob, [_chunk("f", f"def f():\n return {i}\n")], + head_blob_sha=blob) + control_gauges = frequency_gauges(fresh_v, fresh_m) + assert control_gauges.occupancies == 3 and control_gauges.plane_atoms == 3 + assert control_gauges.dedup_factor == pytest.approx(1.0) + assert control_gauges.embeds_avoided == 0 + + +def test_the_rank_frequency_histogram_renders_over_lifetime_n_doc(degenerate: Fixture) -> None: + """D6/§4: the rank-frequency plot of LIFETIME `n_doc(v)`, per lane, checked not assumed. + + The degenerate input is a corpus where every atom sits in exactly one path — the histogram is + then flat and any implementation "renders" it, including one that returns a constant. So the + fixture is asserted to hold a spread (a shared atom at n_doc 2 beside singletons) before the + shape is read.""" + m = degenerate.bench.memberships + + counts = m.n_doc_counts(current_only=False) + # PRECONDITION: the distribution is not flat, so "sorted descending" carries information + assert max(counts.values()) > min(counts.values()), "fixture must hold a frequency spread" + + hist = m.rank_frequency() + assert hist == sorted(hist, reverse=True) # rank i -> the i-th largest n_doc + assert len(hist) == len(counts) == m.occupied_atoms() + assert hist[0] == counts[degenerate.shared_id] == 2 # the shared atom tops the ranking + assert sum(hist) == sum(counts.values()) + + # per lane: L0b holds exactly one atom (the duplicated window), in one path + assert m.rank_frequency(layer=LAYER_CODE_TEXT) == [1] + assert len(m.rank_frequency(layer=LAYER_CODE_AST)) == 4 + + # the CURRENT-cut variant is a different reading and is kept separate (D6): the side branch and + # the superseded revision drop out of it, so it is strictly smaller here. + assert sum(m.rank_frequency(current_only=True)) < sum(hist) + + +def test_current_any_is_equivalent_to_a_positive_n_doc_across_the_gauges( + degenerate: Fixture) -> None: + """The carried invariant `current_any(v) ⇔ n_doc(v, t) > 0` (D6/R3), read through the batch + gauge rather than the point query — the two spellings must agree, or the cheap gauge is + reporting something the expensive truth does not.""" + m = degenerate.bench.memberships + batch = m.n_doc_counts(current_only=True) + + # PRECONDITION: the fixture holds BOTH a current and a fully-superseded atom, so the + # equivalence is not being read on an all-current store. + assert any(v > 0 for v in batch.values()) + superseded = [cid for cid in m.n_doc_counts(current_only=False) if batch.get(cid, 0) == 0] + assert superseded, "fixture must hold an atom whose every occupancy is superseded" + + for row in degenerate.bench.vectors.atom_rows(): + cid = str(row["id"]) + assert bool(row["current"]) == (batch.get(cid, 0) > 0) + assert batch.get(cid, 0) == m.n_doc(cid) # batch and point query agree