From e7b0fbb2cb8333baf2240b8ca9e5b5a25d7e57d7 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 17:16:09 -0700 Subject: [PATCH 01/14] Measure and reduce forwarded routing cost, add write capacity qualification --- .github/workflows/write-capacity.yml | 83 ++ crates/cellule-app/PERFORMANCE.md | 2 + .../performance/2026-09-29-write-capacity.md | 97 +++ crates/cellule-app/qualification/entities.py | 189 ++++- crates/cellule-app/qualification/run.sh | 3 +- crates/cellule-app/qualification/scale.py | 29 +- .../qualification/test_entities.py | 124 ++- crates/cellule-app/tests/entities/process.rs | 2 +- .../tests/entities/process/driver.rs | 83 +- .../tests/entities/process/observation.rs | 92 ++- .../cellule-app/tests/performance_fixture.rs | 88 +- crates/cellule-peer-http/Cargo.toml | 2 +- crates/cellule-peer-http/docs/routing.md | 55 ++ crates/cellule-peer-http/src/lib.rs | 167 +++- crates/cellule-peer-http/src/tests.rs | 757 +++++++++++++++++- .../src/cell/actor/requests.rs | 17 + crates/cellule-runtime/src/fleet/telemetry.rs | 29 + crates/cellule-runtime/src/publication/mod.rs | 7 +- .../tests/runtime/lifecycle/durability.rs | 11 +- .../runtime/lifecycle/durability/proofs.rs | 13 + plans/001-forwarded-routing-and-capacity.md | 223 ++++++ plans/002-write-throughput-bottleneck.md | 208 +++++ plans/README.md | 22 + 23 files changed, 2235 insertions(+), 68 deletions(-) create mode 100644 .github/workflows/write-capacity.yml create mode 100644 crates/cellule-app/performance/2026-09-29-write-capacity.md create mode 100644 plans/001-forwarded-routing-and-capacity.md create mode 100644 plans/002-write-throughput-bottleneck.md create mode 100644 plans/README.md diff --git a/.github/workflows/write-capacity.yml b/.github/workflows/write-capacity.yml new file mode 100644 index 0000000..9966e35 --- /dev/null +++ b/.github/workflows/write-capacity.yml @@ -0,0 +1,83 @@ +name: Cell write capacity qualification + +on: + pull_request: + paths: + - ".github/workflows/write-capacity.yml" + - "Cargo.toml" + - "Cargo.lock" + - ".cargo/**" + - "crates/cellule-app/**" + - "crates/cellule-host/**" + - "crates/cellule-runtime/**" + - "crates/cellule-ltx/**" + - "crates/cellule-store/**" + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: cell-write-capacity-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +jobs: + object-proof: + runs-on: ubuntu-24.04 + timeout-minutes: 100 + steps: + - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 + with: + persist-credentials: false + - name: Pin one source and release binary + run: | + set -euo pipefail + build_state="$RUNNER_TEMP/cell-write-capacity-build" + printf 'CELLULE_CAPACITY_BUILD_STATE=%s\n' "$build_state" >> "$GITHUB_ENV" + mkdir -p "$build_state/source" "$build_state/target-linux" "$build_state/evidence" + git archive HEAD | tar -x -C "$build_state/source" + git rev-parse HEAD > "$build_state/evidence/source-revision.txt" + uname -a > "$build_state/evidence/host.txt" + docker info --format '{{json .}}' > "$build_state/evidence/docker-info.json" + python3 -B -m unittest discover \ + -s crates/cellule-app/qualification -p 'test_*.py' + CELLULE_REFERENCE_STATE="$build_state" \ + docker compose -p cell-write-capacity-build \ + -f "$build_state/source/crates/cellule-app/qualification/compose.yaml" \ + run --name cell-write-capacity-build build \ + 2>&1 | tee "$build_state/evidence/build.log" + binary=$(find "$build_state/target-linux/release/deps" -maxdepth 1 \ + -type f -executable -name 'integration-*') + test -n "$binary" + test "$(printf '%s\n' "$binary" | wc -l)" -eq 1 + sha256sum "$binary" > "$build_state/evidence/binary.sha256" + - name: Run three fresh object-proof capacity repeats + run: | + set -euo pipefail + for repeat in 1 2 3; do + run_state="$RUNNER_TEMP/cell-write-capacity-run-$repeat" + mkdir -p "$run_state/evidence" + ln -s "$CELLULE_CAPACITY_BUILD_STATE/source" "$run_state/source" + ln -s "$CELLULE_CAPACITY_BUILD_STATE/target-linux" "$run_state/target-linux" + cp "$CELLULE_CAPACITY_BUILD_STATE/evidence/source-revision.txt" "$run_state/evidence/" + cp "$CELLULE_CAPACITY_BUILD_STATE/evidence/host.txt" "$run_state/evidence/" + cp "$CELLULE_CAPACITY_BUILD_STATE/evidence/docker-info.json" "$run_state/evidence/" + chmod 1777 "$run_state/evidence" + python3 "$run_state/source/crates/cellule-app/qualification/scale.py" \ + --state "$run_state" --project "cell-write-capacity-r$repeat" \ + --workload capacity + done + - name: Retain provider image identity + if: always() + run: | + docker image ls -a --digests --no-trunc \ + > "$CELLULE_CAPACITY_BUILD_STATE/evidence/images.txt" + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 + if: always() + with: + name: cell-write-capacity-${{ github.run_id }}-${{ github.run_attempt }} + path: | + ${{ runner.temp }}/cell-write-capacity-build/evidence/ + ${{ runner.temp }}/cell-write-capacity-run-*/evidence/ + if-no-files-found: error + retention-days: 7 diff --git a/crates/cellule-app/PERFORMANCE.md b/crates/cellule-app/PERFORMANCE.md index c8ee6a5..b41d310 100644 --- a/crates/cellule-app/PERFORMANCE.md +++ b/crates/cellule-app/PERFORMANCE.md @@ -13,6 +13,7 @@ and [qualification runner](qualification/run.sh) for a fresh run. | Local typed action | [Quickstart](../../docs/quickstart.md) | Write, publish, read-back. | | Three-process RustFS | [`qualification/run.sh`](qualification/run.sh) | Local and forwarded gateway calls, receipts, drained sessions. | | Entity fleet and scaling | [`qualification/entities.py`](qualification/entities.py), [`scale.py`](qualification/scale.py) | Isolated Cell ledgers and bounded traffic. | +| Fixed 12-Cell write capacity | [`qualification/scale.py`](qualification/scale.py) with `--workload capacity` | Hot, uniform, and skewed offered-rate ramps with receipt and overload checks. | | Reader and rollout variants | [Application integration suite](tests/integration.rs) | Selection, replacement, and recovered receipts. | ```mermaid @@ -38,6 +39,7 @@ runs do not establish production capacity. | What did the public action and deployment probes show? | [Action](performance/2026-09-25-public-host-action.md), [RustFS](performance/2026-09-25-public-host-rustfs.md), [three-node Compose](performance/2026-09-27-three-node-compose.md), [replica Compose](performance/2026-09-27-replica-compose.md) | | How were reader recruitment and publication evaluated? | [Recruitment](performance/2026-09-27-reader-recruitment.md), [publication](performance/2026-09-27-reader-publication.md), [expiry](performance/2026-09-27-reader-expiry.md), [loss during load](performance/2026-09-27-reader-loss-during-load.md) | | What were the scale and mixed-load limits? | [Scaling](performance/2026-09-27-reader-scaling.md), [mixed readers](performance/2026-09-27-mixed-readers.md), [host readers Compose](performance/2026-09-27-host-readers-compose.md), [qualification notes](performance/2026-09-27-qualification-notes.md) | +| Is the writable Cell limit established? | [Write capacity measurement status](performance/2026-09-29-write-capacity.md) | ```mermaid flowchart LR diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md new file mode 100644 index 0000000..febdf04 --- /dev/null +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -0,0 +1,97 @@ +# Writable entity capacity: measurement status, 2026-09-29 + +The existing ignored `entity_process_scaling` integration workload schedules +uniform, hot, and skewed traffic at 1, 4, and 16 actions per node per second +across 3, 5, 10, and 20 constrained nodes. Its independent parser checks all +arrivals, stable mutation identities, acknowledged receipts, readback values, +owner records, and final published roots. An integrity pass means that the +reported results are internally consistent; it does not certify a supported +rate when arrivals were missed. + +This revision adds separate raw command-response proof and object-publication +completion records per node. `node-N-responses.tsv` identifies Recorded, +Fleet, or Object as the single proof that released each successful runtime +command response. `node-N-publications.tsv` records queue wait, root +preparation, authority/confirmation time, and total background publication +time per commit sequence. The parser reports their distributions and published +roots/s separately from completed client actions/s. It also reports an +arrival-latency distribution over every scheduled action, including rejected +and late arrivals. The earlier resource, logical object-operation, receipt, +and readback files remain required. + +The new `--workload capacity` selector fixes the fleet at three owners and +12 writable Cells. It schedules 10-second points at 2, 4, 16, 64, 256, and +1024 actions per node per second, stopping each shape at its first overloaded +point. A point is fully served only when every scheduled action succeeds and +admitted work drains within 12 seconds of its 10-second arrival window. The +verifier requires at least one fully served point and one overloaded +point for uniform, hot, and skewed traffic; it rejects missing arrivals, +misstated success, incomplete readback, or an incomplete rate ramp. It reports +the last fully served logical write rate separately from the first overloaded +rate. This profile has not yet been run in the isolated provider environment. + +The dedicated `Cell write capacity qualification` GitHub Actions workflow +builds one release binary and runs three object-proof repeats on a fresh +runner. Each repeat has its own Compose project, RustFS volume, object prefix, +and evidence directory. It uploads raw samples, logs, the binary digest, and +provider details even if a repeat fails. Its output requires review before a +capacity claim; the follower-enabled comparison remains separate. + +After preparing a fresh source/binary snapshot with the Compose qualification +guide, run each repeat with a new state directory and Compose project: + +```sh +python3 "$CELLULE_REFERENCE_STATE/source/crates/cellule-app/qualification/scale.py" \ + --state "$CELLULE_REFERENCE_STATE" --project capacity-unique-run \ + --workload capacity +``` + +The object-proof-only Compose profile records LTX phases and capture timing, +publication objects and bytes, response proof, background publication, +per-Cell root sequence lag, and per-node resource and provider operations. +The resource stream also samples unpublished follower-log bytes from runtime +stats; it is expected to be zero in the object-only lane. +Root lag is derived from the latest acknowledged receipt and completed root +publication at each window's wall-clock end; clock adjustments limit its +precision. A scheduler-late or client-full +overload point identifies the harness admission limit until runtime and +provider phase evidence demonstrates a narrower Cell limiter. + +The existing entity host does not enroll a follower durability lane, so this +profile cannot supply the separate follower-enabled comparison. The added +publication observation aggregates root preparation and provider I/O; it +cannot by itself distinguish predecessor GET/HEAD, immutable PUT, and provider +wait inside that phase. Those limits prevent selecting a safe implementation +change from this revision alone. + +The 2026-09-29 workstation did not provide an isolated provider environment. +The Colima VM had multiple unrelated active RustFS workloads, several above +one CPU, while the Linux qualification binary was building. Running the +3/5/10/20-node profile three times there would mix provider and CPU contention +from those workloads into the Cellule curve. A raw Docker container snapshot +is retained under +`$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/state-20260929/evidence/shared-host-docker-stats.tsv`. +The VM also could not bind-mount the mounted Workspace volume; a disposable +source snapshot was therefore moved under `$HOME/.codex/qualification/` for +the Linux build. No capacity claim or bottleneck classification is made from +that build. + +The earlier disposable snapshot built the Linux `cellule-app` integration test in +release mode with `cargo test --release --locked -p cellule-app --test integration --no-run`. +Its binary SHA-256 is +`6cf7c3dc86ff673c86bcd82b2eedcafbbe2db358f4c16e280d41600476ea9983`. +The snapshot is under `$HOME/.codex/qualification/cellule-capacity-5ca5-state`; +the original source archive and host evidence are under the Workspace path +above. That binary predates the fixed-Cell capacity selector and cannot run +it; create a new source snapshot and release binary for the capacity runs. +This is a compile result, not an execution result. + +Before choosing a write-path optimization, run the workload in a dedicated +provider environment with a new object prefix per repeat. Keep the same +source/binary digest, node limits, Cell count, and arrival schedule across +three repeats; add a separate follower-enabled lane and an offered rate above +the first overload point. Require readback for every acknowledged write and +compare response proof rates with eventual root drain. If root preparation +dominates, collect its GET/HEAD, PUT, provider wait, and CAS split before +proposing a code change. If the three repeats disagree, the result remains +inconclusive. diff --git a/crates/cellule-app/qualification/entities.py b/crates/cellule-app/qualification/entities.py index 38d63dd..15ab2ec 100644 --- a/crates/cellule-app/qualification/entities.py +++ b/crates/cellule-app/qualification/entities.py @@ -10,8 +10,10 @@ STAGES = (3, 5, 10, 20) SHAPES = ("uniform", "hot", "skewed") POINTS = ((1, 4), (4, 16), (16, 64)) +CAPACITY_POINTS = ((2, 8), (4, 16), (16, 64), (64, 128), (256, 256), (1024, 256)) CELLS_PER_NODE = 4 SECONDS = 10 +CAPACITY_DRAIN_GRACE_US = 2_000_000 def rows(path: Path) -> list[dict]: @@ -38,9 +40,101 @@ def distribution(values: list[int]) -> dict: }) +def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dict: + responses = rows(control / f"node-{node}-responses.tsv") + publications = rows(control / f"node-{node}-publications.tsv") + phases = rows(control / f"node-{node}-phases.tsv") + captures = rows(control / f"node-{node}-captures.tsv") + costs = rows(control / f"node-{node}-publication-costs.tsv") + appends = rows(control / f"node-{node}-follower-appends.tsv") + assert responses, f"node {node}: missing command response evidence" + assert publications, f"node {node}: missing publication evidence" + assert phases and captures and costs, f"node {node}: missing LTX or publication phase evidence" + sources = {"Recorded", "Fleet", "Object"} + for row in responses: + assert row["source"] in sources + assert int(row["at_ms"]) > 0 + assert int(row["response_us"]) >= int(row["confirmation_us"]) >= 0 + seen = set() + for row in publications: + key = (row["cell"], int(row["sequence"])) + assert key not in seen, f"node {node}: duplicate publication" + seen.add(key) + assert int(row["at_ms"]) > 0 and int(row["sequence"]) > 0 + assert row["succeeded"] in {"true", "false"} + total = int(row["total_us"]) + assert total >= 0 + for phase in ("queue_wait_us", "preparation_us", "authority_us"): + assert 0 <= int(row[phase]) <= total + for row in phases: + assert int(row["at_ms"]) > 0 and int(row["elapsed_us"]) >= 0 + assert row["succeeded"] in {"true", "false"} + for row in captures: + assert int(row["at_ms"]) > 0 and row["succeeded"] in {"true", "false"} + total = int(row["total_us"]) + assert total >= 0 + for key in ("preparation_us", "schema_check_us", "wal_read_us", "page_collection_us", + "verification_us", "encode_us", "local_write_us", "fsync_us", "checkpoint_us"): + assert 0 <= int(row[key]) <= total + assert int(row["wal_bytes"]) >= 0 and int(row["ltx_bytes"]) >= 0 + for row in costs: + assert int(row["at_ms"]) > 0 and int(row["objects"]) >= 0 and int(row["bytes"]) >= 0 + for row in appends: + assert int(row["at_ms"]) > 0 and int(row["bytes"]) >= 0 + assert row["acknowledged"] in {"true", "false"} + for window in windows: + if node >= window["nodes"]: + continue + start, end = window["started_ms"], window["ended_ms"] + selected_responses = [row for row in responses if start <= int(row["at_ms"]) <= end] + selected_publications = [row for row in publications if start <= int(row["at_ms"]) <= end] + selected_phases = [row for row in phases if start <= int(row["at_ms"]) <= end] + selected_captures = [row for row in captures if start <= int(row["at_ms"]) <= end] + selected_costs = [row for row in costs if start <= int(row["at_ms"]) <= end] + selected_appends = [row for row in appends if start <= int(row["at_ms"]) <= end] + response_sources = {source: sum(row["source"] == source for row in selected_responses) + for source in sorted(sources)} + window.setdefault("node_durability", {})[node] = dict( + response_sources=response_sources, + response_latency=distribution([int(row["response_us"]) for row in selected_responses]), + confirmation_latency=distribution([int(row["confirmation_us"]) for row in selected_responses]), + publication_total=distribution([int(row["total_us"]) for row in selected_publications]), + publication_queue=distribution([int(row["queue_wait_us"]) for row in selected_publications]), + publication_preparation=distribution([int(row["preparation_us"]) for row in selected_publications]), + publication_authority=distribution([int(row["authority_us"]) for row in selected_publications]), + published_roots_per_second=sum(row["succeeded"] == "true" for row in selected_publications) + * 1_000_000 / window["elapsed_us"], + latest_published_sequence_by_cell={cell: max(int(row["sequence"]) for row in publications + if row["cell"] == cell and row["succeeded"] == "true" + and int(row["at_ms"]) <= end) + for cell in {row["cell"] for row in publications + if row["succeeded"] == "true" and int(row["at_ms"]) <= end}}, + failed_publications=sum(row["succeeded"] == "false" for row in selected_publications)) + window["node_durability"][node].update( + ltx_phases={phase: distribution([int(row["elapsed_us"]) for row in selected_phases + if row["phase"] == phase]) + for phase in sorted({row["phase"] for row in selected_phases})}, + capture_total=distribution([int(row["total_us"]) for row in selected_captures]), + uploaded_objects=sum(int(row["objects"]) for row in selected_costs), + uploaded_bytes=sum(int(row["bytes"]) for row in selected_costs), + follower_appends=len(selected_appends), + follower_append_failures=sum(row["acknowledged"] == "false" for row in selected_appends)) + return dict(response_sources={source: sum(row["source"] == source for row in responses) + for source in sorted(sources)}, + response_latency=distribution([int(row["response_us"]) for row in responses]), + publication_total=distribution([int(row["total_us"]) for row in publications]), + capture_total=distribution([int(row["total_us"]) for row in captures]), + uploaded_objects=sum(int(row["objects"]) for row in costs), + uploaded_bytes=sum(int(row["bytes"]) for row in costs), + follower_appends=len(appends), + completed_publications=sum(row["succeeded"] == "true" for row in publications), + failed_publications=sum(row["succeeded"] == "false" for row in publications)) + + def verify_window(control: Path, nodes: int, shape: str, rate_per_node: int, - concurrency: int, window_id: int, positions: dict[int, list[int]]) -> dict: - label = f"entities-{nodes}-{shape}-{rate_per_node}" + concurrency: int, window_id: int, positions: dict[int, list[int]], + prefix: str = "entities") -> dict: + label = f"{prefix}-{nodes}-{shape}-{rate_per_node}" metadata, = rows(control / f"{label}-window.tsv") assert metadata["shape"] == shape for key, expected in dict(window_id=window_id, nodes=nodes, rate_per_node=rate_per_node, @@ -53,7 +147,7 @@ def verify_window(control: Path, nodes: int, shape: str, rate_per_node: int, planned = rate * SECONDS samples = sorted(rows(control / f"{label}.tsv"), key=lambda row: int(row["arrival"])) assert [int(row["arrival"]) for row in samples] == list(range(planned)), "missing or duplicate arrival" - successes, arrival_latencies, writes = [], [], [0] * nodes + successes, arrival_latencies, scheduled_latencies, writes = [], [], [], [0] * nodes outcomes, intervals = {}, [] new_positions = {entity: [] for entity in range(nodes * CELLS_PER_NODE)} for sample in samples: @@ -69,6 +163,7 @@ def verify_window(control: Path, nodes: int, shape: str, rate_per_node: int, assert (entity, sample["kind"]) == destination(shape, arrival, nodes * CELLS_PER_NODE) assert outcome in {"ok", "resolved", "write_only", "not_started", "absent", "client_full", "scheduler_late", "read_failed"} outcomes[outcome] = outcomes.get(outcome, 0) + 1 + scheduled_latencies.append(started + elapsed - scheduled) if outcome in ("client_full", "scheduler_late"): assert elapsed == sequence == read_sequence == count == 0 if outcome == "scheduler_late": @@ -113,18 +208,47 @@ def verify_window(control: Path, nodes: int, shape: str, rate_per_node: int, return dict(nodes=nodes, shape=shape, rate_per_node=rate_per_node, concurrency=concurrency, planned=planned, outcomes=outcomes, fully_served_arrivals=len(successes) == planned, acknowledged_writes_by_node=writes, completed_actions=len(successes), + latest_write_sequence_by_entity={entity: max(values, default=0) + for entity, values in positions.items()}, + completed_writes_per_second=sum(writes) * 1_000_000 / elapsed_us, completed_per_second=len(successes) * 1_000_000 / elapsed_us, service_latency=distribution(successes), arrival_latency=distribution(arrival_latencies), + scheduled_arrival_latency=distribution(scheduled_latencies), started_ms=int(metadata["started_ms"]), ended_ms=int(metadata["ended_ms"]), started_boot_ms=int(metadata["started_boot_ms"]), ended_boot_ms=int(metadata["ended_boot_ms"]), wall_clock_adjustment_ms=int(metadata["ended_ms"]) - int(metadata["started_ms"]) - elapsed_us / 1000, elapsed_us=elapsed_us, peak_client_inflight=peak) -def verify_entities(control: Path) -> dict: - owners = rows(control / "entity-owners.tsv") +def verify_capacity_windows(control: Path, positions: dict[int, list[int]]) -> list[dict]: + schedule = rows(control / "capacity-windows.tsv") + assert schedule, "missing capacity schedule" + assert [int(row["window_id"]) for row in schedule] == list(range(len(schedule))), "missing capacity window" + windows = [] + for shape in SHAPES: + selected = [row for row in schedule if row["shape"] == shape] + assert 2 <= len(selected) <= len(CAPACITY_POINTS), f"{shape}: incomplete rate ramp" + assert [(int(row["rate_per_node"]), int(row["concurrency"])) for row in selected] == list(CAPACITY_POINTS[:len(selected)]), f"{shape}: rate ramp changed" + assert all(row["fully_served"] == "true" for row in selected[:-1]), f"{shape}: ramp continued after overload" + assert selected[-1]["fully_served"] == "false", f"{shape}: missing overload point" + for row in selected: + result = verify_window(control, 3, shape, int(row["rate_per_node"]), + int(row["concurrency"]), int(row["window_id"]), positions, + prefix="capacity") + result["fully_served_window"] = (result["fully_served_arrivals"] and + result["elapsed_us"] <= SECONDS * 1_000_000 + CAPACITY_DRAIN_GRACE_US) + assert result["fully_served_window"] == (row["fully_served"] == "true"), "mislabeled fully served rate" + windows.append(result) + assert [window["shape"] for window in windows] == [row["shape"] for row in schedule], "capacity shape order changed" + return windows + + +def verify_entities(control: Path, capacity: bool = False) -> dict: + stages = (3,) if capacity else STAGES + evidence_prefix = "capacity" if capacity else "entity" + owners = rows(control / f"{evidence_prefix}-owners.tsv") identity = {} - for stage in STAGES: + for stage in stages: selected = [row for row in owners if int(row["stage"]) == stage] assert [int(row["entity"]) for row in selected] == list(range(stage * CELLS_PER_NODE)) for row in selected: @@ -132,15 +256,18 @@ def verify_entities(control: Path) -> dict: assert int(row["owner"]) == entity // CELLS_PER_NODE value = (row["cell"], row["owner"], row["epoch"], row["incarnation"]) assert identity.setdefault(entity, value) == value, "existing Cell ownership changed" - ingress = list(map(int, (control / f"entity-ingress-{stage}.txt").read_text().split())) + ingress = list(map(int, (control / f"{evidence_prefix}-ingress-{stage}.txt").read_text().split())) assert len(ingress) == stage and min(ingress) > 0 and max(ingress) - min(ingress) <= 1 - assert len({value[0] for value in identity.values()}) == 80, "entity targets collapsed" + assert len({value[0] for value in identity.values()}) == stages[-1] * CELLS_PER_NODE, "entity targets collapsed" windows, positions = [], {} - for nodes in STAGES: - for shape in SHAPES: - for rate, concurrency in POINTS: - windows.append(verify_window(control, nodes, shape, rate, concurrency, len(windows), positions)) - roots = rows(control / f"entity-roots-{nodes}.tsv") + for nodes in stages: + if capacity: + windows.extend(verify_capacity_windows(control, positions)) + else: + for shape in SHAPES: + for rate, concurrency in POINTS: + windows.append(verify_window(control, nodes, shape, rate, concurrency, len(windows), positions)) + roots = rows(control / f"{evidence_prefix}-roots-{nodes}.tsv") assert [int(row["entity"]) for row in roots] == list(range(nodes * CELLS_PER_NODE)) for row in roots: entity = int(row["entity"]) @@ -150,14 +277,15 @@ def verify_entities(control: Path) -> dict: for node in range(nodes): assert sum(window["acknowledged_writes_by_node"][node] for window in windows if window["nodes"] == nodes) > 0 resources = {} - for node in range(20): + for node in range(stages[-1]): samples = rows(control / f"node-{node}-resources.tsv") assert len(samples) >= 2 assert all(int(row["active_cells"]) == CELLS_PER_NODE for row in samples) + assert all(int(row["unpublished_node_log_bytes"]) >= 0 for row in samples) for column in ("boot_ms", "cpu_usage_us", "throttled_us", "object_started", "object_finished", "bytes_read", "bytes_written"): values = [int(row[column]) for row in samples] assert values == sorted(values), f"node {node}: {column} regressed" - assert {int(row["stage"]) for row in samples} >= {stage for stage in STAGES if node < stage} + assert {int(row["stage"]) for row in samples} >= {stage for stage in stages if node < stage} local, forwarded = map(int, (control / f"node-{node}.counts").read_text().split()) assert local > 0 and forwarded > 0 observations = rows(control / f"node-{node}-objects.tsv") @@ -170,6 +298,7 @@ def verify_entities(control: Path) -> dict: max_disk_file_bytes=max(int(row["disk_bytes"]) for row in samples), gateway_local=local, gateway_forwarded=forwarded, object_wait=distribution(waits), logical_object_operations=sum(int(row["count"]) for row in observations)) + resources[node]["durability"] = verify_timing_evidence(control, node, windows) for window in windows: if node >= window["nodes"]: continue @@ -183,9 +312,37 @@ def verify_entities(control: Path) -> dict: logical_object_started=int(last["object_started"]) - int(first["object_started"]), logical_object_finished=int(last["object_finished"]) - int(first["object_finished"]), max_memory_current_bytes=max(int(row["memory_current_bytes"]) for row in observed), + max_unpublished_node_log_bytes=max(int(row["unpublished_node_log_bytes"]) for row in observed), max_disk_file_bytes=max(int(row["disk_bytes"]) for row in observed), max_worker_jobs=max(int(row["worker_jobs"]) for row in observed)) + extra = {} + if capacity: + for window in windows: + root_lags = {} + for entity, acknowledged in window["latest_write_sequence_by_entity"].items(): + cell = identity[entity][0] + owner = entity // CELLS_PER_NODE + published = window["node_durability"][owner]["latest_published_sequence_by_cell"].get(cell, 0) + root_lags[entity] = max(0, acknowledged - published) + window["root_lag_commits_by_entity"] = root_lags + window["max_root_lag_commits"] = max(root_lags.values(), default=0) + capacity_curves = {} + for shape in SHAPES: + selected = [window for window in windows if window["shape"] == shape] + fully_served = selected[-2] + overloaded = selected[-1] + capacity_curves[shape] = dict( + max_fully_served_rate_per_node=fully_served["rate_per_node"], + max_fully_served_logical_writes_per_second=fully_served["completed_writes_per_second"], + first_overloaded_rate_per_node=overloaded["rate_per_node"], + first_overloaded_outcomes=overloaded["outcomes"], + max_root_lag_commits_at_fully_served_rate=fully_served["max_root_lag_commits"], + max_root_lag_commits_at_overload=overloaded["max_root_lag_commits"], + published_roots_per_second=sum( + node["published_roots_per_second"] for node in fully_served["node_durability"].values()), + ) + extra["capacity_curves"] = capacity_curves return dict(integrity_verified=True, windows=windows, resources=resources, verified_cells=len(positions), acknowledged_writes=sum(map(len, positions.values())), raw_sha256={path.name: hashlib.sha256(path.read_bytes()).hexdigest() - for path in sorted(control.glob("*.tsv"))}) + for path in sorted(control.glob("*.tsv"))}, **extra) diff --git a/crates/cellule-app/qualification/run.sh b/crates/cellule-app/qualification/run.sh index 9ca9944..3c45eba 100644 --- a/crates/cellule-app/qualification/run.sh +++ b/crates/cellule-app/qualification/run.sh @@ -8,7 +8,8 @@ case "${1:-}" in entities) role=entities; selected=entities::hosts::entity_ledgers_are_isolated_across_three_rustfs_hosts ;; entity-node) role="node-${CELLULE_PERF_PROCESS_NODE:?}"; selected=entities::process::entity_process_node ;; entity-scale) role=driver; selected=entities::process::driver::entity_process_scaling ;; - *) printf 'usage: run.sh node|driver|scale|rollout|entities|entity-node|entity-scale\n' >&2; exit 2 ;; + entity-capacity) role=driver; selected=entities::process::driver::entity_process_capacity ;; + *) printf 'usage: run.sh node|driver|scale|rollout|entities|entity-node|entity-scale|entity-capacity\n' >&2; exit 2 ;; esac binary= for candidate in /target/release/deps/integration-*; do diff --git a/crates/cellule-app/qualification/scale.py b/crates/cellule-app/qualification/scale.py index 6c83f61..04d1eae 100644 --- a/crates/cellule-app/qualification/scale.py +++ b/crates/cellule-app/qualification/scale.py @@ -126,7 +126,7 @@ def verify_reader_loss(control: Path, killed_node: int) -> dict: class Fleet: def __init__(self, state: Path, project: str, overrides: list[Path], workload: str = "readers"): - assert workload in ("readers", "entities") + assert workload in ("readers", "entities", "capacity") self.workload = workload self.state = state.resolve(strict=True) self.project = project @@ -136,7 +136,8 @@ def __init__(self, state: Path, project: str, overrides: list[Path], workload: s existing = self.run("docker", "ps", "-aq", "--filter", f"label=com.docker.compose.project={project}") if existing.strip(): raise ValueError("project already has containers; retain it and choose a fresh project") - self.evidence = self.state / "evidence" / ("scaling" if workload == "readers" else "entity-scaling") + evidence_name = {"readers": "scaling", "entities": "entity-scaling", "capacity": "entity-capacity"}[workload] + self.evidence = self.state / "evidence" / evidence_name self.evidence.mkdir(mode=0o1777) self.evidence.chmod(0o1777) self.control = self.evidence / "control" @@ -150,18 +151,22 @@ def __init__(self, state: Path, project: str, overrides: list[Path], workload: s node = config["services"]["node-0"] # Twenty live nodes plus one killed boot. New boots have new identities; # restarting a deterministic fixture session would bypass expiry fencing. - for index in range(3, 21): + for index in range(3, 3 if workload == "capacity" else 21): added = copy.deepcopy(node) added["environment"].update( CELLULE_PERF_PROCESS_NODE=str(index), CELLULE_PERF_PROCESS_ADVERTISE=f"node-{index}:8080", ) config["services"][f"node-{index}"] = added - config["services"]["driver"]["command"] = ["scale" if workload == "readers" else "entity-scale"] - if workload == "entities": + config["services"]["driver"]["command"] = [{"readers": "scale", "entities": "entity-scale", "capacity": "entity-capacity"}[workload]] + if workload in ("entities", "capacity"): for name, service in config["services"].items(): if name.startswith("node-"): service["command"] = ["entity-node"] + if workload == "capacity": + for name, service in config["services"].items(): + if name.startswith("node-") or name == "driver": + service["environment"]["CELLULE_PERF_PROCESS_ROOT"] = f"capacity-{project}" for service in config["services"].values(): for volume in service.get("volumes", []): if volume["target"] == "/evidence": @@ -270,9 +275,10 @@ def execute(self) -> None: self.verify() def verify(self) -> None: - assert len(self.active) == 20 and len(self.killed) == (1 if self.workload == "readers" else 0) + stages = (3,) if self.workload == "capacity" else (3, 5, 10, 20) + assert len(self.active) == stages[-1] and len(self.killed) == (1 if self.workload == "readers" else 0) assert [(event["action"], event["argument"]) for event in self.events if event["action"] == "scale"] == [ - ("scale", 3), ("scale", 5), ("scale", 10), ("scale", 20) + ("scale", stage) for stage in stages ] roles = {f"node-{node}": container for node, container in self.active.items()} roles["driver"] = self.driver @@ -302,11 +308,12 @@ def verify(self) -> None: binaries.add((self.evidence / f"{role}-binary.sha256").read_text().split()[0]) assert len(binaries) == 1 source = (self.state / "evidence/source-revision.txt").read_text().strip() - if self.workload == "entities": + if self.workload in ("entities", "capacity"): result = dict(workload=self.workload, source=source, binary_sha256=binaries.pop(), - roles=reports, events=self.events, **verify_entities(self.control)) + roles=reports, events=self.events, + **verify_entities(self.control, capacity=self.workload == "capacity")) (self.evidence / "verification.json").write_text(json.dumps(result, indent=2) + "\n") - print(f"Verified entity integrity and resources at 3/5/10/20 nodes; evidence: {self.evidence}", flush=True) + print(f"Verified entity integrity and resources at {stages} nodes; evidence: {self.evidence}", flush=True) return killed = next(iter(self.killed)) driver_log = (self.evidence / "driver.log").read_text() @@ -338,7 +345,7 @@ def main() -> None: parser.add_argument("--state", type=Path, required=True, help="prepared source, binary and evidence directory") parser.add_argument("--project", required=True, help="fresh Compose project; stopped containers are retained") parser.add_argument("--compose-file", type=Path, action="append", default=[], help="explicit image/cache override") - parser.add_argument("--workload", choices=("readers", "entities"), default="readers", help="reader replacement or scheduled writable entity traffic") + parser.add_argument("--workload", choices=("readers", "entities", "capacity"), default="readers", help="reader replacement, scaling, or fixed 12-Cell capacity traffic") args = parser.parse_args() fleet = Fleet(args.state, args.project, args.compose_file, args.workload) try: diff --git a/crates/cellule-app/qualification/test_entities.py b/crates/cellule-app/qualification/test_entities.py index 28035d2..0fde703 100644 --- a/crates/cellule-app/qualification/test_entities.py +++ b/crates/cellule-app/qualification/test_entities.py @@ -4,8 +4,9 @@ from pathlib import Path import tempfile import unittest +from unittest.mock import patch -from entities import verify_window +from entities import destination, verify_capacity_windows, verify_timing_evidence, verify_window class EntityWindowEvidence(unittest.TestCase): @@ -75,5 +76,126 @@ def test_missed_arrival_is_not_fully_served(self): self.assertEqual((result["fully_served_arrivals"], result["completed_actions"]), (False, 29)) +class EntityTimingEvidence(unittest.TestCase): + def setUp(self): + temporary = tempfile.TemporaryDirectory() + self.addCleanup(temporary.cleanup) + self.root = Path(temporary.name) + (self.root / "node-0-responses.tsv").write_text( + "at_ms\tsource\tresponse_us\tconfirmation_us\n" + "100001\tFleet\t1200\t900\n100002\tObject\t2000\t1800\n") + (self.root / "node-0-publications.tsv").write_text( + "at_ms\tcell\tsequence\tqueue_wait_us\tpreparation_us\tauthority_us\ttotal_us\tsucceeded\n" + "100003\tcell-1\t1\t100\t1000\t200\t1500\ttrue\n") + (self.root / "node-0-phases.tsv").write_text( + "at_ms\tphase\telapsed_us\tsucceeded\n100003\tCapture\t700\ttrue\n") + (self.root / "node-0-captures.tsv").write_text( + "at_ms\ttotal_us\tpreparation_us\tschema_check_us\twal_read_us\tpage_collection_us\tverification_us\tencode_us\tlocal_write_us\tfsync_us\tcheckpoint_us\twal_bytes\tltx_bytes\tsucceeded\n" + "100003\t700\t100\t20\t100\t100\t50\t50\t50\t100\t50\t4096\t2048\ttrue\n") + (self.root / "node-0-publication-costs.tsv").write_text( + "at_ms\tobjects\tbytes\n100003\t2\t2048\n") + (self.root / "node-0-follower-appends.tsv").write_text( + "at_ms\tacknowledged\tbytes\n") + self.windows = [dict(nodes=3, started_ms=100000, ended_ms=110000, elapsed_us=10_000_000)] + + def test_response_winner_and_later_publication_are_separate(self): + report = verify_timing_evidence(self.root, 0, self.windows) + self.assertEqual(report["response_sources"], dict(Fleet=1, Object=1, Recorded=0)) + self.assertEqual(self.windows[0]["node_durability"][0]["published_roots_per_second"], 0.1) + self.assertEqual(self.windows[0]["node_durability"][0]["uploaded_objects"], 2) + + def test_incomplete_timing_evidence_is_rejected(self): + (self.root / "node-0-publications.tsv").write_text( + "at_ms\tcell\tsequence\tqueue_wait_us\tpreparation_us\tauthority_us\ttotal_us\tsucceeded\n") + with self.assertRaisesRegex(AssertionError, "missing publication evidence"): + verify_timing_evidence(self.root, 0, self.windows) + + def test_duplicate_publication_is_rejected(self): + path = self.root / "node-0-publications.tsv" + lines = path.read_text().splitlines() + path.write_text("\n".join(lines + [lines[-1]]) + "\n") + with self.assertRaisesRegex(AssertionError, "duplicate publication"): + verify_timing_evidence(self.root, 0, self.windows) + + +class CapacityScheduleEvidence(unittest.TestCase): + def setUp(self): + temporary = tempfile.TemporaryDirectory() + self.addCleanup(temporary.cleanup) + self.root = Path(temporary.name) + self.schedule = self.root / "capacity-windows.tsv" + self.write_schedule([ + (0, "uniform", 2, 8, "true"), (1, "uniform", 4, 16, "false"), + (2, "hot", 2, 8, "true"), (3, "hot", 4, 16, "false"), + (4, "skewed", 2, 8, "true"), (5, "skewed", 4, 16, "false"), + ]) + + def write_schedule(self, entries): + self.schedule.write_text( + "window_id\tshape\trate_per_node\tconcurrency\tfully_served\n" + + "".join("\t".join(map(str, entry)) + "\n" for entry in entries)) + + def write_complete_windows(self): + counts = [0] * 12 + sequences = [1] * 12 + for window_id, shape, rate, concurrency, _ in [ + (0, "uniform", 2, 8, True), (1, "uniform", 4, 16, False), + (2, "hot", 2, 8, True), (3, "hot", 4, 16, False), + (4, "skewed", 2, 8, True), (5, "skewed", 4, 16, False), + ]: + label = f"capacity-3-{shape}-{rate}" + started_ms = 100000 + window_id * 20000 + (self.root / f"{label}-window.tsv").write_text( + "window_id\tnodes\tshape\trate_per_node\tconcurrency\tseconds\tstarted_ms\tended_ms\tstarted_boot_ms\tended_boot_ms\telapsed_us\n" + f"{window_id}\t3\t{shape}\t{rate}\t{concurrency}\t10\t{started_ms}\t{started_ms + 10000}\t{started_ms}\t{started_ms + 10000}\t10000000\n") + samples = ["arrival\tscheduled_us\tstarted_us\telapsed_us\tentity\tkind\toutcome\tsequence\tread_sequence\tcount"] + planned = 30 * rate + for arrival in range(planned): + entity, kind = destination(shape, arrival, 12) + scheduled = arrival * 1_000_000 // (3 * rate) + if rate == 4 and arrival == planned - 1: + samples.append(f"{arrival}\t{scheduled}\t10000000\t0\t{entity}\t{kind}\tscheduler_late\t0\t0\t0") + continue + sequence = 0 + if kind == "write": + counts[entity] += 1 + sequences[entity] += 2 + sequence = sequences[entity] + samples.append(f"{arrival}\t{scheduled}\t{scheduled + 10}\t1000\t{entity}\t{kind}\tok\t{sequence}\t{sequences[entity]}\t{counts[entity]}") + (self.root / f"{label}.tsv").write_text("\n".join(samples) + "\n") + (self.root / f"{label}-readback.tsv").write_text( + "entity\texpected\tactual\tsequence\n" + + "".join(f"{entity}\t{counts[entity]}\t{counts[entity]}\t{sequences[entity]}\n" + for entity in range(12))) + + def test_complete_report_requires_all_shapes_and_overload(self): + self.write_complete_windows() + windows = verify_capacity_windows(self.root, {}) + self.assertEqual(len(windows), 6) + self.assertEqual([window["fully_served_window"] for window in windows], [True, False] * 3) + + def test_completed_arrivals_with_slow_drain_are_not_supported(self): + self.write_complete_windows() + path = self.root / "capacity-3-uniform-2-window.tsv" + path.write_text(path.read_text().replace("10000000\n", "12100000\n")) + with self.assertRaisesRegex(AssertionError, "mislabeled fully served rate"): + verify_capacity_windows(self.root, {}) + + def test_incomplete_report_is_rejected(self): + self.write_schedule([(0, "uniform", 1, 4, "true")]) + with self.assertRaisesRegex(AssertionError, "incomplete rate ramp"): + verify_capacity_windows(self.root, {}) + + def test_rate_mislabeled_as_fully_served_is_rejected(self): + self.write_schedule([ + (0, "uniform", 2, 8, "true"), (1, "uniform", 4, 16, "false"), + (2, "hot", 2, 8, "true"), (3, "hot", 4, 16, "false"), + (4, "skewed", 2, 8, "true"), (5, "skewed", 4, 16, "false"), + ]) + with patch("entities.verify_window", return_value=dict(fully_served_arrivals=False)): + with self.assertRaisesRegex(AssertionError, "mislabeled fully served rate"): + verify_capacity_windows(self.root, {}) + + if __name__ == "__main__": unittest.main() diff --git a/crates/cellule-app/tests/entities/process.rs b/crates/cellule-app/tests/entities/process.rs index 66db8f6..72d8e55 100644 --- a/crates/cellule-app/tests/entities/process.rs +++ b/crates/cellule-app/tests/entities/process.rs @@ -164,7 +164,7 @@ async fn entity_process_node() { observations.sample(stage, &host, directory.path(), &storage, &stats); assert_eq!(host.stats().active_cells(), ENTITIES_PER_NODE); host.shutdown().await.unwrap(); - observations.finish(&storage, &durability.object_waits()); + observations.finish(&storage, &durability); let (local, forwarded) = stats.counts(); assert!(local > 0 && forwarded > 0); publish_marker( diff --git a/crates/cellule-app/tests/entities/process/driver.rs b/crates/cellule-app/tests/entities/process/driver.rs index 1949beb..6ef3b7f 100644 --- a/crates/cellule-app/tests/entities/process/driver.rs +++ b/crates/cellule-app/tests/entities/process/driver.rs @@ -14,9 +14,11 @@ use std::{ use tokio::task::JoinSet; const WINDOW_SECONDS: usize = 10; +const CAPACITY_DRAIN_GRACE_US: u64 = 2_000_000; struct Window { id: usize, + prefix: &'static str, nodes: usize, shape: &'static str, rate_per_node: usize, @@ -26,8 +28,8 @@ struct Window { impl Window { fn label(&self) -> String { format!( - "entities-{}-{}-{}", - self.nodes, self.shape, self.rate_per_node + "{}-{}-{}-{}", + self.prefix, self.nodes, self.shape, self.rate_per_node ) } } @@ -48,6 +50,16 @@ struct Sample { #[tokio::test(flavor = "multi_thread", worker_threads = 4)] #[ignore = "Compose controller required for scheduled entity traffic on 3/5/10/20 nodes"] async fn entity_process_scaling() { + run_entity_process(false).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "Compose controller required for fixed-Cell scheduled capacity traffic"] +async fn entity_process_capacity() { + run_entity_process(true).await; +} + +async fn run_entity_process(capacity: bool) { let sync = env::var("CELLULE_PERF_PROCESS_SYNC").unwrap(); let sync = Path::new(&sync); let mut controller = Controller::new(sync); @@ -58,11 +70,24 @@ async fn entity_process_scaling() { *ApplicationId::from_bytes([82; 16]).as_bytes(), ); let authority = CellAuthority::new(layout.clone()); - let mut owners = BufWriter::new(File::create(sync.join("entity-owners.tsv")).unwrap()); + let evidence_prefix = if capacity { "capacity" } else { "entity" }; + let window_prefix = if capacity { "capacity" } else { "entities" }; + let stages: &[usize] = if capacity { &[3] } else { &[3, 5, 10, 20] }; + let mut owners = + BufWriter::new(File::create(sync.join(format!("{evidence_prefix}-owners.tsv"))).unwrap()); writeln!(owners, "stage\tentity\tcell\towner\tepoch\tincarnation").unwrap(); let mut expected = Vec::new(); let mut window_id = 0; - for nodes in [3, 5, 10, 20] { + let mut capacity_windows = capacity.then(|| { + let mut output = BufWriter::new(File::create(sync.join("capacity-windows.tsv")).unwrap()); + writeln!( + output, + "window_id\tshape\trate_per_node\tconcurrency\tfully_served" + ) + .unwrap(); + output + }); + for &nodes in stages { assert_eq!( controller.command("scale", nodes).await, (0..nodes).collect::>() @@ -114,20 +139,53 @@ async fn entity_process_scaling() { } owners.flush().unwrap(); for shape in ["uniform", "hot", "skewed"] { - for (rate_per_node, concurrency) in [(1, 4), (4, 16), (16, 64)] { + let mut served = false; + let mut overloaded = false; + let points: &[(usize, usize)] = if capacity { + &[ + (2, 8), + (4, 16), + (16, 64), + (64, 128), + (256, 256), + (1024, 256), + ] + } else { + &[(1, 4), (4, 16), (16, 64)] + }; + for &(rate_per_node, concurrency) in points { let window = Window { id: window_id, + prefix: window_prefix, nodes, shape, rate_per_node, concurrency, }; - run_window(sync, &window, client.clone(), &mut expected).await; + let fully_served = run_window(sync, &window, client.clone(), &mut expected).await; + if let Some(output) = capacity_windows.as_mut() { + writeln!( + output, + "{window_id}\t{shape}\t{rate_per_node}\t{concurrency}\t{fully_served}" + ) + .unwrap(); + output.flush().unwrap(); + } window_id += 1; + served |= fully_served; + if capacity && !fully_served { + overloaded = true; + break; + } + } + if capacity { + assert!(served, "{shape}: no fully served capacity point"); + assert!(overloaded, "{shape}: rate ramp did not reach overload"); } } - let mut roots = - BufWriter::new(File::create(sync.join(format!("entity-roots-{nodes}.tsv"))).unwrap()); + let mut roots = BufWriter::new( + File::create(sync.join(format!("{evidence_prefix}-roots-{nodes}.tsv"))).unwrap(), + ); writeln!( roots, "entity\tcell\towner\tepoch\tincarnation\troot_sequence\troot_digest" @@ -156,7 +214,7 @@ async fn entity_process_scaling() { assert!(counts.iter().all(|count| *count > 0)); assert!(counts.iter().max().unwrap() - counts.iter().min().unwrap() <= 1); publish_marker( - &sync.join(format!("entity-ingress-{nodes}.txt")), + &sync.join(format!("{evidence_prefix}-ingress-{nodes}.txt")), counts .iter() .map(usize::to_string) @@ -171,7 +229,7 @@ async fn entity_process_scaling() { let _ = server.await; } publish_marker(&sync.join("stop"), []); - for node in 0..20 { + for node in 0..stages[stages.len() - 1] { wait_for_marker(&sync.join(format!("node-{node}.done"))).await; } assert!( @@ -210,7 +268,7 @@ async fn run_window( window: &Window, client: Arc, expected: &mut [u64], -) { +) -> bool { let rate = window.nodes * window.rate_per_node; let planned = rate * WINDOW_SECONDS; let label = window.label(); @@ -308,6 +366,9 @@ async fn run_window( println!( "ENTITY_WINDOW label={label} planned={planned} complete={complete} elapsed_us={elapsed_us}" ); + complete == planned + && (window.prefix != "capacity" + || elapsed_us <= WINDOW_SECONDS as u64 * 1_000_000 + CAPACITY_DRAIN_GRACE_US) } fn retain_sample(output: &mut BufWriter, samples: &mut Vec, sample: Sample) { diff --git a/crates/cellule-app/tests/entities/process/observation.rs b/crates/cellule-app/tests/entities/process/observation.rs index c47378c..0bf6d9c 100644 --- a/crates/cellule-app/tests/entities/process/observation.rs +++ b/crates/cellule-app/tests/entities/process/observation.rs @@ -1,6 +1,7 @@ //! Cumulative provider operations and per-node Linux resource samples. use super::*; +use crate::performance_fixture::DurabilityRecorder; use cellule_store::{StorageObservation, StorageObserver, StorageOperation, StorageOutcome}; use std::{ fs::File, @@ -37,6 +38,12 @@ pub(super) struct NodeObservations { resources: BufWriter, objects: BufWriter, durability: BufWriter, + responses: BufWriter, + publications: BufWriter, + phases: BufWriter, + captures: BufWriter, + publication_costs: BufWriter, + follower_appends: BufWriter, } impl NodeObservations { @@ -45,11 +52,17 @@ impl NodeObservations { BufWriter::new(File::create(sync.join(format!("node-{node}-{name}.tsv"))).unwrap()) }; let mut resources = create("resources"); - writeln!(resources, "at_ms\tboot_ms\tstage\tactive_cells\tworker_jobs\tprimitive_jobs\thydration_jobs\tretained_bytes\tdisk_reserved_bytes\tdisk_bytes\tcpu_usage_us\tthrottled_us\tmemory_current_bytes\tmemory_peak_bytes\tgateway_local\tgateway_forwarded\tobject_started\tobject_finished\tbytes_read\tbytes_written").unwrap(); + writeln!(resources, "at_ms\tboot_ms\tstage\tactive_cells\tworker_jobs\tprimitive_jobs\thydration_jobs\tretained_bytes\tunpublished_node_log_bytes\tdisk_reserved_bytes\tdisk_bytes\tcpu_usage_us\tthrottled_us\tmemory_current_bytes\tmemory_peak_bytes\tgateway_local\tgateway_forwarded\tobject_started\tobject_finished\tbytes_read\tbytes_written").unwrap(); Self { resources, objects: create("objects"), durability: create("durability"), + responses: create("responses"), + publications: create("publications"), + phases: create("phases"), + captures: create("captures"), + publication_costs: create("publication-costs"), + follower_appends: create("follower-appends"), } } @@ -88,6 +101,7 @@ impl NodeObservations { stats.primitive_jobs() as u64, stats.hydration_jobs() as u64, stats.retained_bytes() as u64, + stats.unpublished_node_log_bytes(), stats.local_disk_reserved_bytes(), disk_bytes(root), cpu_value("usage_usec"), @@ -110,7 +124,7 @@ impl NodeObservations { self.resources.flush().unwrap(); } - pub(super) fn finish(&mut self, storage: &StorageCounters, waits: &[Duration]) { + pub(super) fn finish(&mut self, storage: &StorageCounters, durability: &DurabilityRecorder) { writeln!(self.objects, "operation\toutcome\tcount").unwrap(); for operation in StorageOperation::ALL { for outcome in StorageOutcome::ALL { @@ -126,12 +140,84 @@ impl NodeObservations { } } writeln!(self.durability, "object_wait_us").unwrap(); + let waits = durability.object_waits(); assert!(!waits.is_empty()); - for wait in waits { + for wait in &waits { writeln!(self.durability, "{}", wait.as_micros()).unwrap(); } + writeln!( + self.responses, + "at_ms\tsource\tresponse_us\tconfirmation_us" + ) + .unwrap(); + for (at_ms, source, elapsed, confirmation) in durability.responses() { + writeln!( + self.responses, + "{at_ms}\t{source:?}\t{}\t{}", + elapsed.as_micros(), + confirmation.as_micros() + ) + .unwrap(); + } + writeln!(self.publications, "at_ms\tcell\tsequence\tqueue_wait_us\tpreparation_us\tauthority_us\ttotal_us\tsucceeded").unwrap(); + for (at_ms, cell, timing) in durability.publications() { + writeln!( + self.publications, + "{at_ms}\t{cell:?}\t{}\t{}\t{}\t{}\t{}\t{}", + timing.commit_sequence, + timing.queue_wait.as_micros(), + timing.preparation.as_micros(), + timing.authority.as_micros(), + timing.total.as_micros(), + timing.succeeded + ) + .unwrap(); + } + writeln!(self.phases, "at_ms\tphase\telapsed_us\tsucceeded").unwrap(); + for (at_ms, phase, elapsed, succeeded) in durability.phases() { + writeln!( + self.phases, + "{at_ms}\t{phase:?}\t{}\t{succeeded}", + elapsed.as_micros() + ) + .unwrap(); + } + writeln!(self.captures, "at_ms\ttotal_us\tpreparation_us\tschema_check_us\twal_read_us\tpage_collection_us\tverification_us\tencode_us\tlocal_write_us\tfsync_us\tcheckpoint_us\twal_bytes\tltx_bytes\tsucceeded").unwrap(); + for (at_ms, timing, succeeded) in durability.captures() { + writeln!( + self.captures, + "{at_ms}\t{}\t{}\t{}\t{}\t{}\t{}\t{}\t{}\t{}\t{}\t{}\t{}\t{succeeded}", + timing.total_nanos / 1000, + timing.preparation_nanos / 1000, + timing.schema_check_nanos / 1000, + timing.wal_read_nanos / 1000, + timing.page_collection_nanos / 1000, + timing.verification_nanos / 1000, + timing.encode_nanos / 1000, + timing.local_write_nanos / 1000, + timing.fsync_nanos / 1000, + timing.checkpoint_nanos / 1000, + timing.wal_bytes, + timing.ltx_bytes + ) + .unwrap(); + } + writeln!(self.publication_costs, "at_ms\tobjects\tbytes").unwrap(); + for (at_ms, objects, bytes) in durability.publication_costs() { + writeln!(self.publication_costs, "{at_ms}\t{objects}\t{bytes}").unwrap(); + } + writeln!(self.follower_appends, "at_ms\tacknowledged\tbytes").unwrap(); + for (at_ms, acknowledged, bytes) in durability.follower_appends() { + writeln!(self.follower_appends, "{at_ms}\t{acknowledged}\t{bytes}").unwrap(); + } self.objects.flush().unwrap(); self.durability.flush().unwrap(); + self.responses.flush().unwrap(); + self.publications.flush().unwrap(); + self.phases.flush().unwrap(); + self.captures.flush().unwrap(); + self.publication_costs.flush().unwrap(); + self.follower_appends.flush().unwrap(); } } diff --git a/crates/cellule-app/tests/performance_fixture.rs b/crates/cellule-app/tests/performance_fixture.rs index 4e1f7c7..b893da2 100644 --- a/crates/cellule-app/tests/performance_fixture.rs +++ b/crates/cellule-app/tests/performance_fixture.rs @@ -8,7 +8,8 @@ use std::{ }; use cellule_host::{CellNode, CellNodeBuilder}; -use cellule_runtime::fleet::telemetry::CellTelemetry; +use cellule_ltx::{CaptureTiming, LtxPhase}; +use cellule_runtime::fleet::telemetry::{CellTelemetry, CommandResponseSource, PublicationTiming}; use cellule_runtime::node::lease::NodeLeaseGuard; use cellule_runtime::node::log::DurabilitySource; use cellule_runtime::peer::{ @@ -144,22 +145,101 @@ pub(super) struct PerfFixture { } #[derive(Default)] -pub(super) struct DurabilityRecorder(Mutex>); +pub(super) struct DurabilityRecorder { + proofs: Mutex>, + responses: Mutex>, + publications: Mutex>, + phases: Mutex>, + captures: Mutex>, + publication_costs: Mutex>, + follower_appends: Mutex>, +} impl DurabilityRecorder { pub(super) fn object_waits(&self) -> Vec { - self.0 + self.proofs .lock() .unwrap() .iter() .filter_map(|(source, waited)| (*source == DurabilitySource::Object).then_some(*waited)) .collect() } + + pub(super) fn responses(&self) -> Vec<(i64, CommandResponseSource, Duration, Duration)> { + self.responses.lock().unwrap().clone() + } + + pub(super) fn publications(&self) -> Vec<(i64, cellule_runtime::CellId, PublicationTiming)> { + self.publications.lock().unwrap().clone() + } + + pub(super) fn phases(&self) -> Vec<(i64, LtxPhase, Duration, bool)> { + self.phases.lock().unwrap().clone() + } + + pub(super) fn captures(&self) -> Vec<(i64, CaptureTiming, bool)> { + self.captures.lock().unwrap().clone() + } + + pub(super) fn publication_costs(&self) -> Vec<(i64, u64, u64)> { + self.publication_costs.lock().unwrap().clone() + } + + pub(super) fn follower_appends(&self) -> Vec<(i64, bool, u64)> { + self.follower_appends.lock().unwrap().clone() + } } impl CellTelemetry for DurabilityRecorder { fn durability_proof(&self, source: DurabilitySource, waited: Duration) { - self.0.lock().unwrap().push((source, waited)); + self.proofs.lock().unwrap().push((source, waited)); + } + + fn command_response( + &self, + source: CommandResponseSource, + elapsed: Duration, + confirmation: Duration, + ) { + self.responses + .lock() + .unwrap() + .push((now_ms(), source, elapsed, confirmation)); + } + + fn publication_completed(&self, cell: cellule_runtime::CellId, timing: PublicationTiming) { + self.publications + .lock() + .unwrap() + .push((now_ms(), cell, timing)); + } + + fn ltx_phase(&self, phase: LtxPhase, elapsed: Duration, succeeded: bool) { + self.phases + .lock() + .unwrap() + .push((now_ms(), phase, elapsed, succeeded)); + } + + fn ltx_capture(&self, timing: &CaptureTiming, succeeded: bool) { + self.captures + .lock() + .unwrap() + .push((now_ms(), *timing, succeeded)); + } + + fn publication_cost(&self, objects: u64, bytes: u64) { + self.publication_costs + .lock() + .unwrap() + .push((now_ms(), objects, bytes)); + } + + fn node_log_append(&self, acknowledged: bool, bytes: u64) { + self.follower_appends + .lock() + .unwrap() + .push((now_ms(), acknowledged, bytes)); } } diff --git a/crates/cellule-peer-http/Cargo.toml b/crates/cellule-peer-http/Cargo.toml index 5d18ba4..59acb36 100644 --- a/crates/cellule-peer-http/Cargo.toml +++ b/crates/cellule-peer-http/Cargo.toml @@ -25,6 +25,6 @@ url = "2" x509-cert = "0.2.5" [dev-dependencies] -cellule-store.workspace = true +cellule-store = { workspace = true, features = ["test-support"] } object_store.workspace = true tokio = { workspace = true, features = ["macros", "rt-multi-thread"] } diff --git a/crates/cellule-peer-http/docs/routing.md b/crates/cellule-peer-http/docs/routing.md index e0bdd35..caebe2d 100644 --- a/crates/cellule-peer-http/docs/routing.md +++ b/crates/cellule-peer-http/docs/routing.md @@ -4,6 +4,14 @@ envelope, and bounds request and response bytes. Both owner-routed and direct-node calls reject oversized requests and zero deadlines before lookup. +The sender shares a private, 4,096-entry owner hint across its clones. A hint +is populated only after exact control and signed node-record validation. It +expires after at most five seconds, and at least one second before the signed +node lease. A proven not-started refusal drops the attempted session and +forces one exact refresh within the original deadline. An invalid or lost +response remains an unknown outcome and is never resent blindly. The receiver +still authorizes the peer and fences stale owners. + | Response | Meaning | | --- | --- | | Valid reply | Return the exact peer result. | @@ -27,3 +35,50 @@ sequenceDiagram The same timeout governs resolution, sending, and pacing. A delay that consumes the budget returns a deadline error without sleeping. Other invalid results are never guessed to have failed before execution. + +## Local adapter comparison, 2026-09-29 + +Run the ignored `tests::owner_lookup_performance` benchmark exactly once per +invocation after checking its selector with `-- --list`: + +```sh +CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-route-5ca5" \ + cargo test -p cellule-peer-http tests::owner_lookup_performance --locked \ + -- --ignored --exact --nocapture +``` + +The baseline used `dc387a8` in an isolated source snapshot with only the +benchmark fixture and its test dependency added. The candidate used the same +revision plus the worktree's owner-hint change. Both used a validated signed +owner advertisement, a counting in-memory object store, and local mTLS Axum +peer. Each lane has 1,024 requests at concurrency 1 and 16. The full-adapter +lane calls `send_inner`; the lookup and HTTP lanes isolate its components. +The server returns an encoded peer reply without running receiver dispatch or +checking the request envelope. The raw TSVs and binary digests are retained under +`$HOME/Workspace/crabbuild-target/cellule-route-5ca5/evidence`. + +| Pair | Full adapter p95, c=1 baseline → candidate | p99, c=1 | p95, c=16 | p99, c=16 | +| --- | ---: | ---: | ---: | ---: | +| 1 | 5.874 → 1.887 ms | 8.378 → 3.714 ms | 38.966 → 9.543 ms | 68.701 → 11.669 ms | +| 2 | 9.395 → 2.215 ms | 18.612 → 4.689 ms | 56.168 → 21.710 ms | 82.897 → 35.607 ms | +| 3 | 12.864 → 4.798 ms | 34.210 → 7.957 ms | 92.130 → 25.913 ms | 128.453 → 38.042 ms | + +All three final pairs cleared the 10% p95 gate at both concurrency levels, +improved p99, and raised fully successful concurrency-16 adapter throughput +from 245–409 to 1,106–2,547 requests/s. Warm full-adapter sender reads fell +from two body reads per request to zero while the hint remained live. One of +six candidate warm lanes had two reads across 1,024 calls because its +five-second hint expired during the measurement. Cold sends still made two +metadata reads in both builds. Earlier exploratory runs, including one with +a candidate p99 network spike, remain in the evidence directory; this shared +workstation is not a production latency profile. Local HTTP/TLS is now the +dominant measured adapter phase. This fixture does not measure product ingress, +RustFS, a remote network, or owner SQL/publication time. + +The embedding application maintainer should check whether its ingress route +shares the sender hint and passes a known description through +`with_observed_description`. A product comparison should schedule the same +local and forwarded actions across hot and many-Cell stages, retain owner-loss +and receipt evidence, and count object-store requests per action. Write +throughput attribution is tracked separately by the entity workload and +[`plans/002-write-throughput-bottleneck.md`](../../../plans/002-write-throughput-bottleneck.md). diff --git a/crates/cellule-peer-http/src/lib.rs b/crates/cellule-peer-http/src/lib.rs index a94bfbc..d7a8358 100644 --- a/crates/cellule-peer-http/src/lib.rs +++ b/crates/cellule-peer-http/src/lib.rs @@ -5,7 +5,7 @@ mod tls; pub use tls::{LoadedPeerTls, PeerTlsClient, PeerTlsIdentity, PeerTlsListener, TlsError}; use std::{ - collections::VecDeque, + collections::{HashMap, VecDeque}, future::Future, pin::Pin, sync::{Arc, Mutex}, @@ -15,7 +15,7 @@ use std::{ use cellule_runtime::Error as CellError; use cellule_runtime::cell::application::ApplicationIdentity; use cellule_runtime::control::authority::CellAuthority; -use cellule_runtime::identity::{CellTarget, Digest, SessionId}; +use cellule_runtime::identity::{CellId, CellTarget, Digest, SessionId}; use cellule_runtime::node::{NodeAdvertisement, NodeDirectory}; use cellule_runtime::peer::{PeerRoundTrip, wire as peer_wire}; use futures_util::StreamExt; @@ -23,6 +23,9 @@ use http::{StatusCode, header}; const PEER_FORWARD_PATH: &str = "internal/cells/v1/forward"; const MAX_PEER_CLIENTS: usize = 1_024; +const MAX_OWNER_HINTS: usize = 4_096; +const OWNER_HINT_LIFETIME: Duration = Duration::from_secs(5); +const OWNER_LEASE_MARGIN_MS: i64 = 1_000; /// Content type accepted by the private peer forwarding endpoint. pub const PROTOBUF_MEDIA_TYPE: &str = "application/x-protobuf"; @@ -61,6 +64,7 @@ pub struct PeerHttpRoundTrip { tls: Arc, session: SessionId, clients: Arc>>, + owners: Arc>>, } impl PeerHttpRoundTrip { @@ -80,6 +84,7 @@ impl PeerHttpRoundTrip { tls, session, clients: Arc::new(Mutex::new(VecDeque::new())), + owners: Arc::new(Mutex::new(HashMap::new())), } } @@ -106,23 +111,34 @@ impl PeerHttpRoundTrip { } } let remaining_ms = remaining_timeout(started, timeout_ms)?; - let owner = tokio::time::timeout( - Duration::from_millis(u64::from(remaining_ms)), - self.owner(&target), - ) - .await - .map_err(|_| CellError::Deadline)??; + let owner = + tokio::time::timeout(Duration::from_millis(u64::from(remaining_ms)), async { + if last_retry.is_some() { + self.refresh_owner(&target).await + } else { + self.owner(&target).await + } + }) + .await + .map_err(|_| CellError::Deadline)??; let remaining_ms = remaining_timeout(started, timeout_ms)?; match self.send_once(&owner, request.clone(), remaining_ms).await { Ok(PeerHttpAttempt::Reply(reply)) => return Ok(reply), - Ok(PeerHttpAttempt::Retry(error, delay)) => last_retry = Some((error, delay)), + Ok(PeerHttpAttempt::Retry(error, delay)) => { + self.invalidate_owner(target.cell_id(), owner.session); + last_retry = Some((error, delay)); + } Ok(PeerHttpAttempt::Unknown(error)) => { + self.invalidate_owner(target.cell_id(), owner.session); return Err(CellError::PeerTransportUnknown { context: "peer HTTP response was lost or invalid", source: Box::new(error), }); } - Err(error) => return Err(error), + Err(error) => { + self.invalidate_owner(target.cell_id(), owner.session); + return Err(error); + } } } Err(last_retry.map_or(CellError::CellNotActive, |(error, _)| error)) @@ -162,6 +178,20 @@ impl PeerHttpRoundTrip { async fn owner(&self, target: &CellTarget) -> cellule_runtime::Result { self.scope.check_target(target)?; + let now = now_ms()?; + if let Some(owner) = self.cached_owner(target.cell_id(), now) { + return Ok(owner); + } + self.load_owner(target).await + } + + async fn refresh_owner(&self, target: &CellTarget) -> cellule_runtime::Result { + self.scope.check_target(target)?; + self.load_owner(target).await + } + + async fn load_owner(&self, target: &CellTarget) -> cellule_runtime::Result { + let lookup_started = Instant::now(); let control = self .authority .load(target.cell_id()) @@ -187,12 +217,116 @@ impl PeerHttpRoundTrip { "Cell owner endpoint is not enrolled", )); } - Ok(RemotePeer { + let remote = RemotePeer { session: owner.session, endpoint: url::Url::parse(advertisement.endpoint()).map_err(peer_transport)?, certificate: advertisement.certificate(), public_key: advertisement.verifying_key()?.to_bytes(), - }) + }; + self.remember_owner( + target.cell_id(), + remote.clone(), + advertisement.expires_at_ms(), + now_ms, + lookup_started, + ); + Ok(remote) + } + + fn cached_owner(&self, cell: CellId, now_ms: i64) -> Option { + let Ok(mut owners) = self.owners.lock() else { + return None; + }; + let observed = owners.get(&cell)?; + if observed.expires_at <= Instant::now() + || observed.lease_expires_at_ms <= now_ms.saturating_add(OWNER_LEASE_MARGIN_MS) + { + owners.remove(&cell); + return None; + } + observed.owner.clone() + } + + fn remember_owner( + &self, + cell: CellId, + owner: RemotePeer, + lease_expires_at_ms: i64, + observed_at_ms: i64, + lookup_started: Instant, + ) { + let lease_margin = lease_expires_at_ms + .saturating_sub(observed_at_ms) + .saturating_sub(OWNER_LEASE_MARGIN_MS); + let Ok(lease_margin) = u64::try_from(lease_margin) else { + return; + }; + if lease_margin == 0 { + return; + } + let lifetime = OWNER_HINT_LIFETIME.min(Duration::from_millis(lease_margin)); + let Some(expires_at) = Instant::now().checked_add(lifetime) else { + return; + }; + let Ok(mut owners) = self.owners.lock() else { + return; + }; + if owners + .get(&cell) + .is_some_and(|current| current.lookup_started > lookup_started) + { + return; + } + if owners.len() >= MAX_OWNER_HINTS + && !owners.contains_key(&cell) + && let Some(evicted) = owners.keys().next().copied() + { + owners.remove(&evicted); + } + owners.insert( + cell, + CachedOwner { + owner: Some(owner), + expires_at, + lease_expires_at_ms, + lookup_started, + }, + ); + } + + fn invalidate_owner(&self, cell: CellId, session: SessionId) { + let Ok(mut owners) = self.owners.lock() else { + return; + }; + if owners.get(&cell).is_some_and(|cached| { + cached + .owner + .as_ref() + .is_some_and(|owner| owner.session != session) + }) { + return; + } + let now = Instant::now(); + let Some(expires_at) = now.checked_add(OWNER_HINT_LIFETIME) else { + return; + }; + if owners.len() >= MAX_OWNER_HINTS + && !owners.contains_key(&cell) + && let Some(evicted) = owners.keys().next().copied() + { + owners.remove(&evicted); + } + // Keep a short tombstone so a lookup started before this refusal + // cannot repopulate the invalidated session after the lock is released. + owners.insert( + cell, + CachedOwner { + owner: None, + expires_at, + lease_expires_at_ms: i64::MAX, + lookup_started: now, + }, + ); } fn client(&self, owner: &RemotePeer) -> cellule_runtime::Result { @@ -377,10 +511,12 @@ impl Clone for PeerHttpRoundTrip { tls: Arc::clone(&self.tls), session: self.session, clients: Arc::clone(&self.clients), + owners: Arc::clone(&self.owners), } } } +#[derive(Clone)] struct RemotePeer { session: SessionId, endpoint: url::Url, @@ -388,6 +524,13 @@ struct RemotePeer { public_key: [u8; 32], } +struct CachedOwner { + owner: Option, + expires_at: Instant, + lease_expires_at_ms: i64, + lookup_started: Instant, +} + struct CachedPeerClient { session: SessionId, certificate: Digest, diff --git a/crates/cellule-peer-http/src/tests.rs b/crates/cellule-peer-http/src/tests.rs index d9826a7..efa4722 100644 --- a/crates/cellule-peer-http/src/tests.rs +++ b/crates/cellule-peer-http/src/tests.rs @@ -1,9 +1,13 @@ use super::*; use axum::{Router, body::Body, response::Response, routing::post}; +use cellule_runtime::cell::catalog::{CatalogEntry, CatalogRole, CellCatalog}; +use cellule_runtime::control::{ControlState, Owner, Transition}; +use cellule_runtime::identity::IncarnationId; use cellule_runtime::identity::{ApplicationId, NamespaceId, NodeId, TenantId}; use cellule_runtime::ltx::CellStorageLayout; use cellule_runtime::node::{NodeCapacity, NodeFailureDomain}; use cellule_store::Store; +use cellule_store::test_support::CountingObjectStore; use ed25519_dalek::SigningKey; use object_store::{memory::InMemory, path::Path}; @@ -101,7 +105,10 @@ async fn both_routes_reject_exhausted_deadlines_before_dispatch() { } } -async fn http_attempt(status: StatusCode, delay: Option<&'static str>) -> PeerHttpAttempt { +async fn http_attempt( + status: StatusCode, + delay: Option<&'static str>, +) -> cellule_runtime::Result { let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); let address = listener.local_addr().unwrap(); let router = Router::new().route( @@ -133,7 +140,7 @@ async fn http_attempt(status: StatusCode, delay: Option<&'static str>) -> PeerHt let result = transport.send_once(&peer, vec![1], 5_000).await; stop.send(()).unwrap(); server.await.unwrap(); - result.unwrap() + result } #[tokio::test] @@ -143,7 +150,7 @@ async fn admission_responses_preserve_retry_delay() { StatusCode::SERVICE_UNAVAILABLE, ] { assert!(matches!( - http_attempt(status, Some("2")).await, + http_attempt(status, Some("2")).await.unwrap(), PeerHttpAttempt::Retry(CellError::Capacity(_), delay) if delay == Duration::from_secs(2) )); } @@ -153,8 +160,750 @@ async fn admission_responses_preserve_retry_delay() { async fn server_failure_or_invalid_success_remains_unknown() { for status in [StatusCode::INTERNAL_SERVER_ERROR, StatusCode::OK] { assert!(matches!( - http_attempt(status, None).await, + http_attempt(status, None).await.unwrap(), PeerHttpAttempt::Unknown(_) )); } } + +#[tokio::test] +async fn authentication_refusal_is_not_an_owner_retry() { + for status in [StatusCode::UNAUTHORIZED, StatusCode::FORBIDDEN] { + assert!(matches!( + http_attempt(status, None).await, + Err(CellError::PeerAuthorization(_)) + )); + } +} + +async fn owner_lookup_fixture() -> (PeerHttpRoundTrip, CellTarget, Arc) { + owner_lookup_fixture_with_endpoint( + "https://owner.example:443".into(), + Digest::from_bytes([23; 32]), + Digest::from_bytes([23; 32]), + SigningKey::from_bytes(&[27; 32]), + Arc::new(TestClients), + ) + .await +} + +async fn owner_lookup_fixture_with_endpoint( + endpoint: String, + certificate: Digest, + fleet: Digest, + signing_key: SigningKey, + clients: Arc, +) -> (PeerHttpRoundTrip, CellTarget, Arc) { + let application = ApplicationId::from_bytes([21; 16]); + let tenant = TenantId::from_bytes([22; 16]); + let digest = Digest::from_bytes([23; 32]); + let counted = Arc::new(CountingObjectStore::new(Arc::new(InMemory::new()))); + let layout = CellStorageLayout::new( + Store::new(counted.clone()), + Path::from("owner-lookup-performance"), + *application.as_bytes(), + ); + let target = CellTarget::new( + tenant, + application, + NamespaceId::from_bytes([24; 16]), + b"partition", + ) + .unwrap(); + let catalog = CellCatalog::new(layout.clone(), tenant); + let proof = catalog + .provision(CatalogEntry::new(&target, CatalogRole::Sql, digest, 1).unwrap()) + .await + .unwrap(); + let authority = CellAuthority::new(layout.clone()); + let directory = NodeDirectory::new(layout, fleet, digest, digest); + let now = now_ms().unwrap(); + let advertisement = NodeAdvertisement::sign( + NodeId::from_bytes([25; 16]), + SessionId::from_bytes([26; 16]), + endpoint, + fleet, + certificate, + digest, + digest, + &signing_key, + 1, + now, + now + 30_000, + vec![digest], + vec![1], + NodeFailureDomain::default(), + NodeCapacity::default(), + ) + .unwrap(); + directory.create(advertisement.clone(), now).await.unwrap(); + authority + .create_initial( + &proof, + IncarnationId::from_bytes([28; 16]), + Owner { + session: advertisement.session(), + endpoint: advertisement.endpoint().to_owned(), + }, + ) + .await + .unwrap(); + counted.reset(); + let transport = PeerHttpRoundTrip::new( + Arc::new(ApplicationIdentity::new(tenant, application)), + authority, + directory, + clients, + SessionId::from_bytes([29; 16]), + ); + (transport, target, counted) +} + +#[tokio::test] +async fn exact_owner_lookup_reads_control_and_signed_session() { + let (transport, target, counted) = owner_lookup_fixture().await; + let owner = transport.owner(&target).await.unwrap(); + assert_eq!(owner.session, SessionId::from_bytes([26; 16])); + assert_eq!(counted.counts().body_requests(), 2); + transport.owner(&target).await.unwrap(); + assert_eq!(counted.counts().body_requests(), 2); +} + +#[tokio::test] +async fn invalidated_hint_resolves_a_signed_owner_after_authority_takeover() { + let (transport, target, counted) = owner_lookup_fixture().await; + let old = transport.owner(&target).await.unwrap(); + let now = now_ms().unwrap(); + let successor = NodeAdvertisement::sign( + NodeId::from_bytes([35; 16]), + SessionId::from_bytes([36; 16]), + "https://successor.example:443".into(), + Digest::from_bytes([23; 32]), + Digest::from_bytes([23; 32]), + Digest::from_bytes([23; 32]), + Digest::from_bytes([23; 32]), + &SigningKey::from_bytes(&[37; 32]), + 1, + now, + now + 30_000, + vec![Digest::from_bytes([23; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity::default(), + ) + .unwrap(); + transport + .directory + .create(successor.clone(), now) + .await + .unwrap(); + let observed = transport + .authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + let next = observed + .value() + .takeover(Owner { + session: successor.session(), + endpoint: successor.endpoint().to_owned(), + }) + .unwrap(); + transport + .authority + .transition(&observed, next, Transition::Takeover) + .await + .unwrap(); + counted.reset(); + assert_eq!(transport.owner(&target).await.unwrap().session, old.session); + assert_eq!(counted.counts().body_requests(), 0); + transport.invalidate_owner(target.cell_id(), old.session); + assert_eq!( + transport.owner(&target).await.unwrap().session, + successor.session() + ); + assert_eq!(counted.counts().body_requests(), 2); +} + +#[tokio::test] +async fn retired_owner_session_fails_closed_after_hint_invalidation() { + let (transport, target, counted) = owner_lookup_fixture().await; + let owner = transport.owner(&target).await.unwrap(); + let now = now_ms().unwrap(); + let enrolled = transport + .directory + .load(owner.session, now) + .await + .unwrap() + .unwrap(); + transport.directory.withdraw(&enrolled, now).await.unwrap(); + transport.invalidate_owner(target.cell_id(), owner.session); + counted.reset(); + assert!(transport.owner(&target).await.is_err()); + assert_eq!(counted.counts().body_requests(), 2); +} + +#[tokio::test] +async fn tombstoned_cell_is_not_routed_from_an_invalidated_hint() { + let (transport, target, counted) = owner_lookup_fixture().await; + let old = transport.owner(&target).await.unwrap(); + let observed = transport + .authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + let mut next = observed.value().clone(); + next.epoch += 1; + next.revision += 1; + next.progress += 1; + next.state = ControlState::Tombstoned; + next.owner = None; + transport + .authority + .transition(&observed, next, Transition::Tombstone) + .await + .unwrap(); + transport.invalidate_owner(target.cell_id(), old.session); + counted.reset(); + assert!(matches!( + transport.owner(&target).await, + Err(CellError::CellNotActive) + )); + assert_eq!(counted.counts().body_requests(), 1); +} + +#[tokio::test] +async fn owner_hint_is_shared_scoped_and_invalidated_by_session() { + let (transport, target, counted) = owner_lookup_fixture().await; + let owner = transport.owner(&target).await.unwrap(); + let clone = transport.clone(); + clone.owner(&target).await.unwrap(); + assert_eq!(counted.counts().body_requests(), 2); + + let foreign = CellTarget::new( + TenantId::from_bytes([31; 16]), + target.application(), + target.namespace(), + target.partition(), + ) + .unwrap(); + assert!(matches!( + clone.owner(&foreign).await, + Err(CellError::PeerAuthorization(_)) + )); + assert_eq!(counted.counts().body_requests(), 2); + + transport.invalidate_owner(target.cell_id(), SessionId::from_bytes([32; 16])); + clone.owner(&target).await.unwrap(); + assert_eq!(counted.counts().body_requests(), 2); + transport.invalidate_owner(target.cell_id(), owner.session); + clone.owner(&target).await.unwrap(); + assert_eq!(counted.counts().body_requests(), 4); +} + +#[tokio::test] +async fn delayed_old_lookup_cannot_replace_new_owner_hint() { + let (transport, target, _) = owner_lookup_fixture().await; + let old_started = Instant::now(); + let new_started = old_started + Duration::from_millis(1); + let now = now_ms().unwrap(); + let peer = RemotePeer { + session: SessionId::from_bytes([33; 16]), + endpoint: "https://new.example:443".parse().unwrap(), + certificate: Digest::from_bytes([34; 32]), + public_key: [35; 32], + }; + transport.remember_owner( + target.cell_id(), + peer.clone(), + now + 10_000, + now, + new_started, + ); + let old = RemotePeer { + session: SessionId::from_bytes([36; 16]), + ..peer + }; + transport.remember_owner(target.cell_id(), old, now + 10_000, now, old_started); + assert_eq!( + transport.owner(&target).await.unwrap().session, + peer.session + ); + transport.invalidate_owner(target.cell_id(), peer.session); + transport.remember_owner( + target.cell_id(), + RemotePeer { + session: SessionId::from_bytes([36; 16]), + endpoint: "https://old.example:443".parse().unwrap(), + certificate: Digest::from_bytes([34; 32]), + public_key: [35; 32], + }, + now + 10_000, + now, + old_started, + ); + assert!( + transport + .owners + .lock() + .unwrap() + .get(&target.cell_id()) + .unwrap() + .owner + .is_none() + ); +} + +#[tokio::test] +async fn near_expired_owner_lease_is_not_cached() { + let (transport, target, counted) = owner_lookup_fixture().await; + let now = now_ms().unwrap(); + let peer = RemotePeer { + session: SessionId::from_bytes([37; 16]), + endpoint: "https://old.example:443".parse().unwrap(), + certificate: Digest::from_bytes([38; 32]), + public_key: [39; 32], + }; + transport.remember_owner(target.cell_id(), peer, now + 500, now, Instant::now()); + assert_eq!( + transport.owner(&target).await.unwrap().session, + SessionId::from_bytes([26; 16]) + ); + assert_eq!(counted.counts().body_requests(), 2); +} + +#[tokio::test] +async fn expired_owner_hint_forces_exact_lookup_for_concurrent_callers() { + let (transport, target, counted) = owner_lookup_fixture().await; + transport.owner(&target).await.unwrap(); + transport + .owners + .lock() + .unwrap() + .get_mut(&target.cell_id()) + .unwrap() + .expires_at = Instant::now() - Duration::from_millis(1); + counted.reset(); + let calls = (0..5).map(|_| { + let transport = transport.clone(); + let target = target.clone(); + async move { transport.owner(&target).await.unwrap() } + }); + let owners = futures_util::future::join_all(calls).await; + assert!( + owners + .iter() + .all(|owner| owner.session == owners[0].session) + ); + assert!(counted.counts().body_requests() >= 2); + assert!(counted.counts().body_requests() <= 10); +} + +#[tokio::test] +async fn owner_hint_cache_stays_bounded() { + let (transport, target, _) = owner_lookup_fixture().await; + let owner = transport.owner(&target).await.unwrap(); + let now = now_ms().unwrap(); + for index in 0..MAX_OWNER_HINTS + 32 { + let mut bytes = [0; 32]; + bytes[..8].copy_from_slice(&(index as u64).to_be_bytes()); + transport.remember_owner( + CellId::from_bytes(bytes), + owner.clone(), + now + 10_000, + now, + Instant::now(), + ); + } + assert_eq!(transport.owners.lock().unwrap().len(), MAX_OWNER_HINTS); +} + +#[tokio::test] +async fn ambiguous_peer_response_does_not_retry_cached_route() { + use std::sync::atomic::{AtomicUsize, Ordering}; + + let hits = Arc::new(AtomicUsize::new(0)); + let server_hits = Arc::clone(&hits); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let router = Router::new().route( + "/internal/cells/v1/forward", + post(move || { + server_hits.fetch_add(1, Ordering::Relaxed); + async { + Response::builder() + .status(StatusCode::OK) + .body(Body::from("bad")) + .unwrap() + } + }), + ); + let server = tokio::spawn(async move { axum::serve(listener, router).await.unwrap() }); + let (transport, target, counted) = owner_lookup_fixture().await; + let now = now_ms().unwrap(); + let peer = RemotePeer { + session: SessionId::from_bytes([26; 16]), + endpoint: format!("http://{address}/").parse().unwrap(), + certificate: Digest::from_bytes([23; 32]), + public_key: [27; 32], + }; + transport.remember_owner(target.cell_id(), peer, now + 10_000, now, Instant::now()); + assert!(matches!( + transport.send_inner(target.clone(), vec![1], 1_000).await, + Err(CellError::PeerTransportUnknown { .. }) + )); + assert_eq!(hits.load(Ordering::Relaxed), 1); + assert_eq!(counted.counts().body_requests(), 0); + assert!( + transport + .owners + .lock() + .unwrap() + .get(&target.cell_id()) + .unwrap() + .owner + .is_none() + ); + server.abort(); +} + +#[tokio::test] +async fn cancelled_send_releases_owner_hint_for_next_request() { + let started = Arc::new(tokio::sync::Notify::new()); + let release = Arc::new(tokio::sync::Notify::new()); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let router = Router::new().route( + "/internal/cells/v1/forward", + post({ + let started = Arc::clone(&started); + let release = Arc::clone(&release); + move || { + let started = Arc::clone(&started); + let release = Arc::clone(&release); + async move { + started.notify_one(); + release.notified().await; + Response::new(Body::empty()) + } + } + }), + ); + let server = tokio::spawn(async move { axum::serve(listener, router).await.unwrap() }); + let (transport, target, counted) = owner_lookup_fixture().await; + let now = now_ms().unwrap(); + let owner = RemotePeer { + session: SessionId::from_bytes([26; 16]), + endpoint: format!("http://{address}/").parse().unwrap(), + certificate: Digest::from_bytes([23; 32]), + public_key: [27; 32], + }; + transport.remember_owner(target.cell_id(), owner, now + 10_000, now, Instant::now()); + let sender = transport.clone(); + let request_target = target.clone(); + let request = + tokio::spawn(async move { sender.send_inner(request_target, vec![1], 5_000).await }); + started.notified().await; + request.abort(); + assert!(request.await.unwrap_err().is_cancelled()); + assert_eq!( + transport.owner(&target).await.unwrap().session, + SessionId::from_bytes([26; 16]) + ); + assert_eq!(counted.counts().body_requests(), 0); + release.notify_one(); + server.abort(); +} + +#[tokio::test] +async fn not_started_refusal_refreshes_authority_without_resending_to_stale_peer() { + use std::sync::atomic::{AtomicUsize, Ordering}; + + let hits = Arc::new(AtomicUsize::new(0)); + let server_hits = Arc::clone(&hits); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let router = Router::new().route( + "/internal/cells/v1/forward", + post(move || { + server_hits.fetch_add(1, Ordering::Relaxed); + async { + Response::builder() + .status(StatusCode::SERVICE_UNAVAILABLE) + .body(Body::empty()) + .unwrap() + } + }), + ); + let server = tokio::spawn(async move { axum::serve(listener, router).await.unwrap() }); + let (transport, target, counted) = owner_lookup_fixture().await; + let now = now_ms().unwrap(); + let peer = RemotePeer { + session: SessionId::from_bytes([26; 16]), + endpoint: format!("http://{address}/").parse().unwrap(), + certificate: Digest::from_bytes([23; 32]), + public_key: [27; 32], + }; + transport.remember_owner(target.cell_id(), peer, now + 10_000, now, Instant::now()); + assert!( + transport + .send_inner(target.clone(), vec![1], 1_000) + .await + .is_err() + ); + assert_eq!(hits.load(Ordering::Relaxed), 1); + assert_eq!(counted.counts().body_requests(), 2); + server.abort(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "manual adapter owner-lookup baseline"] +async fn owner_lookup_performance() { + use std::time::Instant; + + let mut raw = String::from("lane\tconcurrency\telapsed_us\n"); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let certificate_dir = std::env::temp_dir().join(format!( + "cellule-peer-bench-cert-{}-{}", + std::process::id(), + now_ms().unwrap() + )); + std::fs::create_dir(&certificate_dir).unwrap(); + let openssl = |args: &[&str]| { + let output = std::process::Command::new("openssl") + .args(args) + .current_dir(&certificate_dir) + .output() + .unwrap(); + assert!( + output.status.success(), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + }; + openssl(&[ + "req", + "-x509", + "-newkey", + "ed25519", + "-nodes", + "-keyout", + "ca.key", + "-out", + "ca.crt", + "-subj", + "/CN=Cellule benchmark CA", + "-days", + "1", + "-addext", + "basicConstraints=critical,CA:TRUE", + "-addext", + "keyUsage=critical,keyCertSign,cRLSign", + ]); + openssl(&[ + "req", + "-new", + "-newkey", + "ed25519", + "-nodes", + "-keyout", + "leaf.key", + "-out", + "leaf.csr", + "-subj", + "/CN=localhost", + ]); + std::fs::write(certificate_dir.join("leaf.ext"), "subjectAltName=DNS:localhost\nextendedKeyUsage=serverAuth,clientAuth\nkeyUsage=digitalSignature\n").unwrap(); + openssl(&[ + "x509", + "-req", + "-in", + "leaf.csr", + "-CA", + "ca.crt", + "-CAkey", + "ca.key", + "-CAcreateserial", + "-out", + "leaf.crt", + "-days", + "1", + "-extfile", + "leaf.ext", + ]); + let tls = LoadedPeerTls::load( + &certificate_dir.join("leaf.crt"), + &certificate_dir.join("leaf.key"), + &certificate_dir.join("ca.crt"), + "localhost", + ) + .unwrap(); + let (transport, target, counted) = owner_lookup_fixture_with_endpoint( + format!("https://localhost:{}/", address.port()), + tls.certificate(), + tls.fleet(), + tls.signing_key().clone(), + Arc::new(tls.client_identity()), + ) + .await; + let response = cellule_runtime::peer::encode_peer_reply(&peer_wire::PeerReply { + outcome: Some(peer_wire::peer_reply::Outcome::Read(peer_wire::ReadReply { + receipt: None, + result: Some(peer_wire::read_reply::Result::Description( + peer_wire::CellDescription { + cell_id: target.cell_id().as_bytes().to_vec(), + incarnation: vec![28; 16], + code: vec![23; 32], + schema: 1, + }, + )), + })), + }) + .unwrap(); + let response = Arc::new(response); + let router = Router::new().route( + "/internal/cells/v1/forward", + post(move || { + let response = Arc::clone(&response); + async move { + Response::builder() + .status(StatusCode::OK) + .header(header::CONTENT_TYPE, PROTOBUF_MEDIA_TYPE) + .header(header::CACHE_CONTROL, "no-store") + .body(Body::from(response.as_ref().clone())) + .unwrap() + } + }), + ); + let tls_listener = tls.listener(listener); + let server = tokio::spawn(async move { axum::serve(tls_listener, router).await.unwrap() }); + let peer = Arc::new(RemotePeer { + session: SessionId::from_bytes([26; 16]), + endpoint: format!("https://localhost:{}/", address.port()) + .parse() + .unwrap(), + certificate: tls.certificate(), + public_key: tls.signing_key().verifying_key().to_bytes(), + }); + let cold_before = counted.counts().body_requests(); + let cold_started = Instant::now(); + transport + .send_inner(target.clone(), vec![1], 5_000) + .await + .unwrap(); + let cold_us = cold_started.elapsed().as_micros(); + let cold_reads = counted.counts().body_requests() - cold_before; + raw.push_str(&format!("cold_adapter\t1\t{cold_us}\n")); + println!("PERF cold_adapter elapsed_us={cold_us} reads={cold_reads}"); + for concurrency in [1_usize, 16] { + let before = counted.counts().body_requests(); + let started = Instant::now(); + let mut samples = Vec::with_capacity(1_024); + for _ in 0..(1_024 / concurrency) { + let calls = (0..concurrency).map(|_| { + let transport = transport.clone(); + let target = target.clone(); + async move { + let call = Instant::now(); + transport.owner(&target).await.unwrap(); + call.elapsed().as_micros() + } + }); + samples.extend(futures_util::future::join_all(calls).await); + } + let total = started.elapsed(); + let reads = counted.counts().body_requests() - before; + assert_eq!(samples.len(), 1_024); + assert!(reads <= 2 * samples.len()); + samples.sort_unstable(); + for sample in &samples { + raw.push_str(&format!("owner_lookup\t{concurrency}\t{sample}\n")); + } + println!( + "PERF owner_lookup concurrency={concurrency} calls={} elapsed_ms={:.3} throughput_per_s={:.1} p50_us={} p95_us={} p99_us={} reads={reads}", + samples.len(), + total.as_secs_f64() * 1_000.0, + samples.len() as f64 / total.as_secs_f64(), + samples[samples.len() / 2], + samples[samples.len() * 95 / 100], + samples[samples.len() * 99 / 100], + ); + let started = Instant::now(); + let mut network_samples = Vec::with_capacity(1_024); + for _ in 0..(1_024 / concurrency) { + let calls = (0..concurrency).map(|_| { + let transport = transport.clone(); + let peer = Arc::clone(&peer); + async move { + let call = Instant::now(); + assert!(matches!( + transport.send_once(&peer, vec![1], 5_000).await.unwrap(), + PeerHttpAttempt::Reply(_) + )); + call.elapsed().as_micros() + } + }); + network_samples.extend(futures_util::future::join_all(calls).await); + } + let network_total = started.elapsed(); + network_samples.sort_unstable(); + for sample in &network_samples { + raw.push_str(&format!("peer_http\t{concurrency}\t{sample}\n")); + } + println!( + "PERF peer_http concurrency={concurrency} calls={} elapsed_ms={:.3} throughput_per_s={:.1} p50_us={} p95_us={} p99_us={}", + network_samples.len(), + network_total.as_secs_f64() * 1_000.0, + network_samples.len() as f64 / network_total.as_secs_f64(), + network_samples[network_samples.len() / 2], + network_samples[network_samples.len() * 95 / 100], + network_samples[network_samples.len() * 99 / 100], + ); + // Refresh an expired observation outside the warm lane. Every sample + // below begins with the same live owner hint in the candidate. + transport.owner(&target).await.unwrap(); + let reads_before = counted.counts().body_requests(); + let started = Instant::now(); + let mut full_samples = Vec::with_capacity(1_024); + for _ in 0..(1_024 / concurrency) { + let calls = (0..concurrency).map(|_| { + let transport = transport.clone(); + let target = target.clone(); + async move { + let call = Instant::now(); + transport.send_inner(target, vec![1], 5_000).await.unwrap(); + call.elapsed().as_micros() + } + }); + full_samples.extend(futures_util::future::join_all(calls).await); + } + let full_total = started.elapsed(); + let reads = counted.counts().body_requests() - reads_before; + full_samples.sort_unstable(); + assert_eq!(full_samples.len(), 1_024); + for sample in &full_samples { + raw.push_str(&format!("full_adapter\t{concurrency}\t{sample}\n")); + } + println!( + "PERF full_adapter concurrency={concurrency} calls={} elapsed_ms={:.3} throughput_per_s={:.1} p50_us={} p95_us={} p99_us={} reads={reads}", + full_samples.len(), + full_total.as_secs_f64() * 1_000.0, + full_samples.len() as f64 / full_total.as_secs_f64(), + full_samples[full_samples.len() / 2], + full_samples[full_samples.len() * 95 / 100], + full_samples[full_samples.len() * 99 / 100], + ); + } + server.abort(); + let report = std::env::temp_dir().join(format!( + "cellule-peer-owner-lookup-{}-{}.tsv", + std::process::id(), + now_ms().unwrap() + )); + std::fs::write(&report, raw).unwrap(); + println!("PERF raw={}", report.display()); + std::fs::remove_dir_all(certificate_dir).unwrap(); +} diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index 41c3e3a..f9e2bed 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -371,6 +371,8 @@ pub(super) fn start_publication( pool: &SqlWorkerPool, tasks: &mut JoinSet, ) { + use crate::fleet::telemetry::PublicationTiming; + let Some(mut publisher) = active.publisher.take() else { return; }; @@ -383,6 +385,7 @@ pub(super) fn start_publication( - i128::from(active.published_sequence); let incarnation = active.incarnation; let commit_sequence = publication.pending.outcome().commit_sequence(); + let queue_wait = publication.submitted_at.elapsed(); tracing::debug!( target: "cellule_runtime::action", event = "cell_publication_started", @@ -409,8 +412,11 @@ pub(super) fn start_publication( let mut publication_proof = Some(publication.proof); let fleet_deadline = std::time::Instant::now() + FLEET_PUBLICATION_GRACE; let mut retry_delay = std::time::Duration::from_millis(100); + let mut preparation = std::time::Duration::ZERO; + let mut authority = std::time::Duration::ZERO; let result = async { let expected = publication.pending.outcome().clone(); + let preparation_started = std::time::Instant::now(); let prepared = loop { match publisher.prepare(&publication.pending).await { Ok(prepared) => break prepared, @@ -432,7 +438,9 @@ pub(super) fn start_publication( Err(error) => return Err(error), } }; + preparation = preparation_started.elapsed(); pool.bind_prepared(cell, prepared.clone()).await?; + let authority_started = std::time::Instant::now(); let root = loop { match publisher .publish_prepared(&prepared, publication.pending.next_due_ms()) @@ -457,6 +465,7 @@ pub(super) fn start_publication( Err(error) => return Err(error), } }; + authority = authority_started.elapsed(); if let Some(durability) = publication.durability.as_ref() { durability.prove_object().await?; } else { @@ -471,6 +480,14 @@ pub(super) fn start_publication( Ok(()) } .await; + publisher.record_publication_timing(PublicationTiming { + queue_wait, + preparation, + authority, + total: publication.submitted_at.elapsed(), + succeeded: result.is_ok(), + commit_sequence, + }); tracing::debug!( target: "cellule_runtime::action", event = "cell_publication_completed", diff --git a/crates/cellule-runtime/src/fleet/telemetry.rs b/crates/cellule-runtime/src/fleet/telemetry.rs index 3d7106d..9ce026a 100644 --- a/crates/cellule-runtime/src/fleet/telemetry.rs +++ b/crates/cellule-runtime/src/fleet/telemetry.rs @@ -1,6 +1,7 @@ //! Bounded operational telemetry emitted by the runtime. use std::{sync::Arc, time::Duration}; +use crate::CellId; use crate::fleet::pressure::PressureState; use crate::node::log::DurabilitySource; @@ -15,6 +16,24 @@ pub enum CommandResponseSource { Object, } +/// One completed object publication, which may finish after a follower-proof +/// response has already been released for the same commit sequence. +#[derive(Clone, Copy, Debug)] +pub struct PublicationTiming { + /// Time spent queued behind earlier roots for this Cell. + pub queue_wait: Duration, + /// Time spent preparing the immutable root, including bounded retries. + pub preparation: Duration, + /// Time spent publishing the prepared root through authority CAS. + pub authority: Duration, + /// Time from queued publication to terminal completion. + pub total: Duration, + /// Whether object publication completed and the worker confirmed the root. + pub succeeded: bool, + /// Sequence used to correlate this observation with a request trace. + pub commit_sequence: u64, +} + /// Outcome of an actor-owned resident route lookup. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum ResidentRouteOutcome { @@ -135,6 +154,10 @@ pub trait CellTelemetry: Send + Sync { ) { } + /// Records background root progress separately from the response winner. + /// Cell IDs and sequences are for local trace correlation, never metric labels. + fn publication_completed(&self, _cell: CellId, _timing: PublicationTiming) {} + /// Records how one commit's node-log submission resolved. fn durability_submission(&self, _outcome: DurabilitySubmissionOutcome) {} @@ -235,6 +258,12 @@ impl CellTelemetryHandle { } } + pub(crate) fn publication_completed(&self, cell: CellId, timing: PublicationTiming) { + if let Some(telemetry) = self.inner.get() { + telemetry.publication_completed(cell, timing); + } + } + pub(crate) fn catalog_read(&self, kind: CatalogReadKind, elapsed: Duration, succeeded: bool) { if let Some(telemetry) = self.inner.get() { telemetry.catalog_read(kind, elapsed, succeeded); diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index c729ef8..2791f47 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -2,7 +2,7 @@ use crate::cell::executor::{CellExecutor, StoredOutcome}; use crate::control::Transition; use crate::control::authority::{CellAuthority, VersionedControl}; -use crate::fleet::telemetry::DurabilitySubmissionOutcome; +use crate::fleet::telemetry::{DurabilitySubmissionOutcome, PublicationTiming}; use crate::identity::{ApplicationId, encode_hex}; use crate::node::durability::NodeDurability; use crate::node::log::CommitTicket; @@ -117,6 +117,11 @@ impl CellPublisher { .durability_proof(crate::node::log::DurabilitySource::Object, waited); } + pub(crate) fn record_publication_timing(&self, timing: PublicationTiming) { + self.telemetry + .publication_completed(self.observed.value().cell, timing); + } + /// Returns the control version the publisher last observed. #[must_use] pub fn control(&self) -> &VersionedControl { diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs index 1c389ec..d463533 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs @@ -1,14 +1,17 @@ //! Node-log durability, fleet proofs, and byte admission. use super::*; -use cellule_runtime::fleet::telemetry::{CellTelemetry, CommandResponseSource}; +use cellule_runtime::fleet::telemetry::{CellTelemetry, CommandResponseSource, PublicationTiming}; mod admission; mod proofs; mod recovery; #[derive(Default)] -pub(super) struct RecordingResponses(pub(super) Mutex>); +pub(super) struct RecordingResponses( + pub(super) Mutex>, + pub(super) Mutex>, +); impl CellTelemetry for RecordingResponses { fn command_response( @@ -23,4 +26,8 @@ impl CellTelemetry for RecordingResponses { } self.0.lock().unwrap().push(source); } + + fn publication_completed(&self, _cell: cellule_runtime::CellId, timing: PublicationTiming) { + self.1.lock().unwrap().push(timing); + } } diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs index 4820d95..8356c6b 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs @@ -68,6 +68,12 @@ async fn accepted_control_cas_with_lost_response_releases_once() { CommandResponseSource::Recorded ] ); + { + let publication_timings = responses.1.lock().unwrap(); + assert_eq!(publication_timings.len(), 1); + assert_eq!(publication_timings[0].commit_sequence, 1); + assert!(publication_timings[0].succeeded); + } let restored = fixture._directory.path().join("lost-cas-restored.sqlite"); let verified = fixture.replica.open_root(&root).await.unwrap(); @@ -164,6 +170,7 @@ async fn capture_failure_after_sql_commit_fences_until_authoritative_recovery() ); runtime.shutdown().await.unwrap(); assert_eq!(responses.0.lock().unwrap().as_slice(), &[]); + assert!(responses.1.lock().unwrap().is_empty()); let restored = fixture._directory.path().join("capture-restored.sqlite"); let verified = fixture.replica.open_root(&root).await.unwrap(); @@ -847,6 +854,12 @@ async fn exercise_fleet_ack_drain(lose_response: bool) { CommandResponseSource::Recorded ] ); + { + let publication_timings = responses.1.lock().unwrap(); + assert_eq!(publication_timings.len(), 1); + assert_eq!(publication_timings[0].commit_sequence, 1); + assert!(publication_timings[0].succeeded); + } let verified = fixture.replica.open_root(&published_root).await.unwrap(); let recovered = fixture diff --git a/plans/001-forwarded-routing-and-capacity.md b/plans/001-forwarded-routing-and-capacity.md new file mode 100644 index 0000000..384853b --- /dev/null +++ b/plans/001-forwarded-routing-and-capacity.md @@ -0,0 +1,223 @@ +# Plan 001: Reduce forwarded owner lookup cost and prove the gain + +> **Executor:** Follow the steps in order and preserve raw evidence outside the +> checkout. Do not weaken owner fencing, peer authentication, or the durable +> response gate. If a STOP condition applies, report it before editing further. +> +> **Drift check:** `git diff --stat dc387a8..HEAD -- crates/cellule-peer-http crates/cellule-runtime/src/client` +> If a listed file changed, compare the current-state facts below with live +> code before starting. Re-plan any changed behavior. + +## Status + +- **Priority:** P1 +- **Effort:** L, split into measurement, implementation, and qualification commits +- **Risk:** MED, because a stale route or ambiguous mutation must fail safely +- **Depends on:** none +- **Category:** performance +- **Planned at:** `dc387a8`, 2026-09-29 +- **Execution update, 2026-09-29:** Adapter benchmark, bounded owner hint, + failure tests, and paired comparison completed. The application ingress + follow-up remains with the embedding application maintainer. + +## Why this matters + +The generic non-owner path may read catalog and control at ingress, then read +control and a signed node advertisement in the HTTP peer sender before the +request reaches the owner. The sender repeats its two reads on each request. +Removing those reads from a healthy warm path could lower latency and object +store load, but the benefit must be measured against the peer RTT and owner +execution cost. A route observation is only a destination hint; the receiving +owner still verifies and admits every request. + +## Current state + +- `crates/cellule-runtime/src/client/runtime.rs:53-59,76-92`: the default + `RuntimeCellTransport` resolver loads catalog and exact control before + deciding whether a handle is local. A product may supply another resolver. +- `crates/cellule-peer-http/src/lib.rs:93-128,163-195`: `send_inner` permits + two attempts inside one deadline; `owner` loads exact control and then the + signed directory record on each attempt. `lib.rs:198-225` caches pinned + HTTP clients, not owner observations. +- `crates/cellule-runtime/src/client/mod.rs:618-627` provides + `with_observed_description` so a caller with an authority description can + skip `Describe`. Do not create a second description mechanism. +- `crates/cellule-runtime/src/peer/dispatch/mod.rs:134-153` authorizes a + pre-resolved peer request; actor admission fences a stale owner. +- `crates/cellule-peer-http/src/tls.rs:214-220` uses HTTP/1.1 and an existing + idle connection pool. Change it only if connection evidence calls for it. +- `crates/cellule-app/tests/host.rs` has forwarded action probes, but its + `peer_round_trip` is a custom TCP fixture from `tests/fleet.rs`. It does + **not** exercise `PeerHttpRoundTrip` and cannot measure this adapter change. +- `crates/cellule-runtime/docs/vfs-ltx-scale-plan.md` is a historical design + record. It says an embedding product implemented a bounded owner hint; + inspect that product separately before proposing ingress edits there. + +Follow the root and nearest crate `AGENTS.md`: preserve source errors, use +no `unwrap`, `expect`, or `panic!` outside tests, and keep tests beside the +module. Product HTTP endpoints and authorization are outside Cellule. + +## Scope + +**May modify:** `crates/cellule-peer-http/src/lib.rs`, `src/tests.rs`, +`docs/routing.md`, and a test-only feature declaration in `Cargo.toml` for +the counting-store fixture. A new sibling test module may be added if the +module-layout check requires one. The executor may update `plans/README.md` status. + +**Do not modify:** runtime authority, catalog, actor, peer protobuf, LTX, +follower proof, SQLite durability, read replica policy, product routing, or +resource limits. Do not add a public cache configuration option. + +## Commands + +Use a unique target directory beneath a mounted `$HOME/Workspace/crabbuild-target`. +For this plan, the examples use `cellule-route-001`: + +| Check | Command | Expected | +| --- | --- | --- | +| Adapter tests | `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-route-001" cargo test -p cellule-peer-http --locked` | All pass. | +| Runtime peer tests | `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-route-001" cargo test -p cellule-runtime peer --locked` | All pass. | +| Format | `cargo fmt --all --check` | Exit 0. | +| Lint | `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-route-001" cargo clippy --workspace --all-targets --all-features --locked -- -D warnings` | Exit 0. | +| Layout/boundaries | `python3 scripts/check-module-layout.py` and `python3 scripts/check-boundaries.py` | Both exit 0. | +| Docs/contracts | `python3 scripts/check-doc-rust-fences.py`, `python3 scripts/check-doc-links.py`, `node crates/cellule-runtime/docs/validate.mjs` | All exit 0. | + +Run broad suites and provider/process tests in CI or an isolated snapshot. +Use a disposable store and fresh object prefix. Record revision, binary digest, +limits, workload parameters, raw samples, and failures for every comparison. + +## Steps + +### 1. Establish an adapter-specific baseline + +Create one ignored benchmark test in `cellule-peer-http/src/tests.rs`, modeled +on the existing `fixture`, `TestClients`, and local Axum response fixtures. +Set up a valid published control record and signed owner advertisement using +the existing authority/directory APIs. Use a fixture store that counts exact +control and node-record reads; do not bypass record validation. Return a +valid encoded peer reply. Measure one cold send and at least 1,000 warm sends +at concurrency 1 and 16. Record lookup time, HTTP time, end-to-end adapter +p50/p95/p99, successful requests/s, send attempts, and metadata reads per +logical request. A small non-ignored test must assert the report has all +lanes and cannot silently contain zero samples. + +Run the benchmark three times on `dc387a8` under fixed CPU and network +conditions, saving raw samples outside the checkout. Check `--list` for the +exact ignored test selector before running it, then require one executed test. +Do not present this local adapter result as product ingress or RustFS capacity. + +**Verify:** the adapter test command passes; each of three baseline reports +contains at least 1,000 warm samples in each lane, exact attempted/success +counts, and the expected nonzero control and directory reads. If sender +lookup is below 10% of adapter p95 in all three runs, STOP and report the +dominant measured phase instead of adding a cache. + +### 2. Add a bounded advisory owner observation + +In `cellule-peer-http/src/lib.rs`, share a private owner-observation map +across `PeerHttpRoundTrip` clones. Key by `CellId` only after +`scope.check_target`. Store the owner session, parsed enrolled endpoint, +pinned certificate and public key, insertion time, and signed node expiry. +Populate only after the existing successful control load, directory load, +and endpoint-match check. Cap at 4,096 entries. Expire each observation at +the earlier of insertion plus five seconds and signed advertisement expiry +minus one second. Do not cache absence or authorization errors. + +The first attempt may use a live observation. A proven not-started refusal +invalidates only the attempted session and forces the existing one +authoritative retry under the original deadline. An invalid/lost response +remains an unknown outcome and is never blindly resent. A delayed older +lookup must not replace a newer observation. Hold no cache lock across +provider I/O, TLS construction, or network send. Preserve signed bytes, +peer hop limit, receiver authorization, and actor fencing. + +**Verify:** adapter tests pass. Add a counting-store assertion that two +healthy sends to one Cell cause one sender control read and one directory +read total; an expired hint causes one new lookup; and one stale hint causes +at most one authoritative retry. + +### 3. Prove failure behavior + +Extend `cellule-peer-http/src/tests.rs` using its existing HTTP response +classification fixtures. Cover owner transfer, node expiration or +retirement, endpoint/key change, deletion, five simultaneous expired-hint +callers, 429/503 with and without `Retry-After`, a delayed old lookup, and +response loss after a mutation may have been accepted. Test cancellation +releases cache state and a later request can proceed. Verify that only a +proven not-started outcome is retried and an ambiguous mutation keeps its +unknown-outcome classification. Do not infer correctness from a cache hit. + +**Verify:** adapter and runtime peer test commands pass; instrumented tests +show no resend after unknown outcome, at most two sends on a proven +not-started retry, and no send to a stale session after invalidation. + +### 4. Compare and decide + +Repeat the exact Step 1 adapter benchmark three times on the candidate, +using the same CPU/network limits and request count. Compare cold and warm +p50/p95/p99, successful requests/s at both concurrency levels, control and +directory reads/request, attempts, 503s, and unknown outcomes. Keep raw +samples and per-run paired comparisons. + +**Verify:** warm healthy sends have zero sender control/directory reads. +Keep the cache only if warm adapter p95 improves at least 10% in two of three +paired runs **or** fully successful requests/s at concurrency 16 improves at +least 10%, with no p99 regression beyond 5%, no increase in unknown outcomes +or unexplained 503s, and no increase in metadata reads per action. These +are experiment gates, not a product SLO. If the gate fails, revert the cache +change and retain the benchmark evidence. + +### 5. Record the next end-to-end and throughput work + +Update `cellule-peer-http/docs/routing.md` with the exact benchmark command, +baseline/candidate revisions, paired results, and what the adapter benchmark +does and does not cover. Hand the embedding application maintainer two +specific checks: (1) determine whether ingress shares its bounded owner +hint with the outgoing transport and passes its observed description through +`with_observed_description`; (2) run same-image, fixed-offered-rate local +versus forwarded actions across hot-Cell and many-Cell stages, with scheduled +arrivals, owner loss during load, raw receipt verification, retries, and +object-store requests/action. Inspect that product repo before prescribing +its file edits. The adapter cache alone cannot remove entry-router reads. + +For write throughput, use the existing response-source telemetry and the +phase experiments in `crates/cellule-runtime/docs/vfs-ltx-scale-plan.md` +work packet 4. Attribute queue, SQLite, capture, follower proof, root +preparation, provider wait, and control CAS; identify the largest measured +limiter and create a separate implementation plan for it. Do not change +publication or SQLite policy under this routing plan. + +**Verify:** the routing document contains the paired adapter results, +the remaining dominant adapter phase, and the product/throughput follow-up +owner. The docs/contracts commands above pass. + +## Done criteria + +- [x] Three raw baseline and three raw candidate adapter reports exist outside + the checkout, with cold/warm and concurrency 1/16 samples. +- [x] A warm sender performs zero control/directory reads while its hint is live; miss and + invalidation retain bounded exact reads and the original deadline. +- [x] Unknown mutation outcomes are never blindly replayed; stale-session, + cancellation, authentication, and refusal tests pass. +- [x] The numerical Step 4 gate passes, or the cache is reverted and the + measured reason is documented. +- [x] Focused tests, format, lint, boundaries, layout, and docs/contracts pass. +- [x] Only in-scope source/docs files and `plans/README.md` change. + +## STOP conditions + +- Drift changed the owner-resolution or retry contract. +- The fixture cannot distinguish an ambiguous outcome from a proven + not-started refusal. +- A hint would need to grant ownership, bypass receiver authorization, + extend a signed lease, or relax durable response gating. +- Step 1 shows sender lookup is not a material adapter latency phase. +- A failing verification recurs after one focused repair attempt, or a + required edit lies outside Scope. + +## Maintenance notes + +Review cache invalidation when peer response codes or node advertisement +fields change. Keep hint expiry shorter than the signed session lease. +Connection-pool tuning, read replicas, product ingress routing, and write +publication need their own measured decisions. diff --git a/plans/002-write-throughput-bottleneck.md b/plans/002-write-throughput-bottleneck.md new file mode 100644 index 0000000..77f7d24 --- /dev/null +++ b/plans/002-write-throughput-bottleneck.md @@ -0,0 +1,208 @@ +# Plan 002: Measure the write-throughput limit and select one safe optimization + +> **Executor:** This is a measurement and decision plan. Do not change +> publication, SQLite, or LTX semantics before the bottleneck is demonstrated. +> Keep all raw provider evidence outside the checkout. Stop on an invariant +> failure rather than altering the qualification profile. +> +> **Drift check:** `git diff --stat dc387a8..HEAD -- crates/cellule-runtime/src/cell crates/cellule-runtime/src/publication crates/cellule-runtime/src/fleet/telemetry.rs crates/cellule-ltx/src/replica crates/cellule-app/tests crates/cellule-app/qualification` +> Re-read any changed path against the facts below before editing. + +## Status + +- **Priority:** P1 +- **Effort:** M for attribution and workload evidence; later optimization is a separate plan +- **Risk:** LOW for bounded instrumentation, HIGH for an unmeasured durability change +- **Depends on:** none; compare its results with Plan 001 before prioritizing code +- **Category:** performance +- **Planned at:** `dc387a8`, 2026-09-29 +- **Execution update, 2026-09-29:** Separate response, publication, LTX, + capture, and upload observations are implemented. A fixed 12-Cell capacity + selector and verifier are implemented and locally tested. A dedicated + GitHub Actions workflow is ready to run three isolated object-proof repeats. + Its execution, a follower-enabled lane, and a measured-limiter follow-up + remain. + +## Why this matters + +Removing a forwarding lookup may improve a request but cannot raise a hot +Cell's write rate if its publisher or durability proof is saturated. The +existing response telemetry distinguishes recorded, follower, and object +proofs, while root preparation and SQL/capture costs are not yet tied to the +same action and offered-load curve. This plan establishes the sustainable +rate and identifies the single phase that limits it. + +## Current state and contracts + +- `crates/cellule-runtime/src/cell/actor/requests.rs:289` owns + `prove_command`; `src/cell/actor/admission.rs:146` reports the final + `CommandResponseSource` and elapsed/confirmation time. +- `crates/cellule-runtime/src/fleet/telemetry.rs:115-177` defines the bounded + `CellTelemetry` callbacks for response source, publication cost, LTX phase, + control reads, and admission. New labels must be finite; do not use Cell IDs + or tenant strings. +- `crates/cellule-runtime/src/publication/mod.rs` serializes each Cell's + object-root preparation and authority CAS. A follower proof may release a + response before object publication finishes; completed publication time is + therefore not automatically response latency. +- `crates/cellule-ltx/src/db/mod.rs:520` has `capture_deferred`, the path used + by the runtime; a `capture()` comparison has a different sync barrier. +- `crates/cellule-ltx/src/replica/prepare.rs:448-461,549` prepares immutable + roots and uploads dependencies. The predecessor graph still checks origin + presence, and missing metadata must fail closed. +- `crates/cellule-app/tests/process_scaling.rs:392-480` schedules 300 writes + 200 ms apart across a 60-second mixed-reader window. Its Python verifier in + `qualification/scale.py` requires that schedule. It is a correctness and + 5 writes/s profile, not a maximum-throughput curve. +- `crates/cellule-runtime/docs/vfs-ltx-scale-plan.md` work packet 4 lists + outstanding phase and saturation experiments. Its dated 0.3 ms capture and + 87–139 ms root-preparation figures use different harnesses and cannot be + subtracted to infer end-to-end latency. + +Preserve one fenced writer, ordered receipts, exact-root reconstruction, and +the rule that a successful command follows object publication or a +recoverable follower-log proof. Read root and nearest crate `AGENTS.md` before +changing a crate. + +## Scope + +**May modify:** finite phase instrumentation in +`crates/cellule-runtime/src/cell/actor/requests.rs`, +`src/cell/actor/admission.rs`, `src/fleet/telemetry.rs`, and focused sibling +tests; a benchmark-only workload under `crates/cellule-app/tests/` with its +integration-suite entry; a corresponding parser/test under +`crates/cellule-app/qualification/`; and a dated report under +`crates/cellule-app/performance/`. Add other production instrumentation only +after a scope review names the exact file and callback. + +**Do not modify:** SQL sync mode, LTX format, root validation, authority CAS, +follower quorum, existing qualification schedules, resource ceilings, or +the root's expected evidence. Add a separate workload selector; do not +parameterize the existing 300-write verifier until its contract tests are +updated independently. + +## Commands + +| Check | Command | Expected | +| --- | --- | --- | +| Focused runtime tests | `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-capacity-002" cargo test -p cellule-runtime --features test-support --locked` | Pass. | +| App integration | `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-capacity-002" cargo test -p cellule-app --test integration --locked` | Non-ignored tests pass. | +| Evidence parser | `python3 -m unittest discover -s crates/cellule-app/qualification -p test_scale.py` | Pass. | +| Format and lint | `cargo fmt --all --check` and `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-capacity-002" cargo clippy --workspace --all-targets --all-features --locked -- -D warnings` | Both pass. | +| Boundaries/docs | `python3 scripts/check-boundaries.py`, `python3 scripts/check-module-layout.py`, `python3 scripts/check-doc-rust-fences.py`, `python3 scripts/check-doc-links.py`, `node crates/cellule-runtime/docs/validate.mjs` | All pass. | + +Run broad suites, process faults, and provider runs in CI or an isolated +snapshot. Set `CARGO_TARGET_DIR` under the mounted Workspace volume with a +directory unique to the checkout. Use a disposable provider and new object +prefix for each run. + +## Steps + +### 1. Make response and background publication separately visible + +Add bounded phase observations around queue wait, SQL handler/commit, +`capture_deferred`, follower append/proof, root preparation, provider I/O, +authority CAS, and final confirmation. Each observation must carry only a +fixed phase/outcome label and timing. Tie phases to one request in an +isolated trace or raw benchmark record, without putting request or Cell IDs +in aggregate metric labels. Record separately (a) which proof released the +response and (b) when background object publication drained. A cancellation +or unknown mutation must not be counted as a successful response. + +**Verify:** focused runtime tests pass. Add a unit test that a follower-first +response has exactly one response winner and may have a later object proof; +an object-first response has one object winner; and cancellation produces no +false successful response. Compare event counts with existing committed +sequence assertions. + +### 2. Add a scheduled capacity workload without weakening existing tests + +Add a new ignored application integration workload with fixed Cell count and +recorded arrival times. Use at least three shapes: one hot writable Cell, +many evenly distributed Cells, and skewed Cells. At each shape, run increasing +offered rates until one fully served rate is followed by an overloaded rate. +Record every scheduled, started, completed, rejected, timed-out, and late +arrival; retain one stable mutation identity for any retry. Check each +acknowledged receipt with a minimum-receipt readback. Record owner/epoch, +CPU, memory, open Cells, publisher queue age, unpublished bytes, root lag, +object GET/HEAD/PUT and bytes, attempts, and p50/p95/p99 for successful and +all scheduled actions. The parser must reject missing samples, an inflated +success rate, a missing readback, and a rate mislabeled as fully served. + +Use the existing `process_scaling.rs` scheduled-arrival and +`qualification/scale.py` receipt checks as patterns, but give the new +workload its own selector and evidence filenames. Keep the 300-write mixed +reader workload and its tests unchanged. + +**Verify:** app tests and parser tests pass. A deliberately incomplete +synthetic report must be rejected. A completed report must show both a fully +served point and an explicit overloaded point for every workload shape. + +### 3. Run paired evidence and locate saturation + +In an isolated provider environment, run each shape three times with the +same revision, binary digest, node CPU/memory limits, provider identity, +Cell count, and offered-rate schedule. Run a follower-enabled and an +object-proof-only lane; do not combine their percentiles. Include a cold +activation phase and steady resident phase. Preserve raw samples and logs +outside the checkout, then write a dated summary under +`crates/cellule-app/performance/` linking their locations and checksums. + +Calculate the maximum **fully served** logical writes/s for each shape. +Classify each overloaded point as admission, SQL/queue, capture, follower, +publication, provider, or CPU/memory saturation using its phase and resource +evidence. Report published roots/s and root lag independently of response +throughput. A response released on follower proof does not establish that +the publisher can drain indefinitely. + +**Verify:** all three repeats agree on the dominant phase for each shape, +or the report explicitly says the result is inconclusive. Every successful +write has readback evidence; no run reports a supported rate at an overloaded +point. The report includes p95/p99 and raw evidence checksums. + +### 4. Write the next implementation plan for the measured limiter + +If root preparation dominates, inspect its predecessor GET/HEAD, directory +read, immutable PUT, provider wait, and CAS split before proposing a change. +If queue/SQL dominates, inspect worker occupancy and checkpoint/full-image +events. If follower proof dominates, inspect enrollment and append wait. +If the runs disagree or the provider is saturated externally, repeat the +measurement under an isolated profile instead of changing code. + +Write a new plan under `plans/` for **one** measured limiter. It must name +the exact source paths, unchanged safety invariants, a before/after workload, +and a numeric gate: improved fully served rate or p95/p99 without more +failed proofs, retry pressure, root lag, or object requests per logical write. +Do not implement that follow-up as part of Plan 002. + +**Verify:** the new plan identifies one limiting phase with supporting raw +measurements and a regression test for its corresponding invariant. Update +`plans/README.md` with the next plan and this plan's status. + +## Done criteria + +- [ ] Response-winning proof and later publication are separately measured. +- [ ] A new scheduled workload covers hot, uniform, and skewed Cells without + changing the existing mixed-reader qualification contract. +- [ ] Three repeats per shape distinguish fully served from overloaded rates + and retain raw readback, resource, and object-store evidence. +- [ ] A single measured bottleneck has a follow-up implementation plan, or + the report says precisely why the evidence is inconclusive. +- [ ] Focused tests, parser tests, format, lint, boundaries, and docs checks pass. + +## STOP conditions + +- A successful response can no longer be tied to exactly one durable proof. +- The workload needs to weaken the existing 300-write/60-second profile or + omit scheduled arrivals to pass. +- A phase counter would block the actor or add unbounded-cardinality labels. +- Any acknowledged receipt fails readback or owner-loss recovery. +- The measured limiter cannot be distinguished from shared-host provider or + resource contention after three controlled runs. + +## Maintenance notes + +Review phase labels when the publication pipeline changes. Preserve both +response rate and eventual root-drain rate in future capacity reports. +Never treat a faster local `capture()` benchmark as evidence for the runtime's +`capture_deferred()` path. diff --git a/plans/README.md b/plans/README.md new file mode 100644 index 0000000..c50c23f --- /dev/null +++ b/plans/README.md @@ -0,0 +1,22 @@ +# Executable performance plans + +These plans were prepared against Cellule commit `dc387a8` on 2026-09-29. +Read the whole plan and its STOP conditions before execution. Keep raw provider +and process evidence outside the checkout. + +| Order | Plan | Priority | Effort | Status | +| --- | --- | --- | --- | --- | +| 1 | [Cut forwarded routing work and prove the capacity gain](001-forwarded-routing-and-capacity.md) | P1 | L | DONE: bounded adapter hint and three paired comparisons; product ingress follow-up is external | +| 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | IN PROGRESS: fixed-Cell selector and CI object-proof runner ready; provider evidence and follower lane pending | + +Status values: TODO, IN PROGRESS, DONE, BLOCKED (with reason), REJECTED (with +reason). Update the row after executing the plan. + +## Dependency and scope notes + +Plan 001 begins with measurement and keeps an explicit stop gate before adding +an owner hint. It covers the optional Cellule peer HTTP adapter. The embedding +application owns ingress routing; its integration work requires its own repo. +Plan 002 is independent of Plan 001 and measures the hot-Cell and fleet-wide +write limit. Its output is a separate code-change plan for the measured +limiter; it does not guess a new durability policy in advance. From 80eb571507fe0a5f74ebebaa959ab6fd7ee9a604 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 17:36:43 -0700 Subject: [PATCH 02/14] Attribute capacity runs with per-operation provider timings --- .../performance/2026-09-29-write-capacity.md | 46 +++++++++++++++++-- crates/cellule-app/qualification/entities.py | 34 +++++++++++++- .../qualification/test_entities.py | 15 +++++- .../tests/entities/process/driver.rs | 8 +++- .../tests/entities/process/observation.rs | 27 ++++++++++- plans/002-write-throughput-bottleneck.md | 9 ++-- plans/README.md | 2 +- 7 files changed, 128 insertions(+), 13 deletions(-) diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index febdf04..6656743 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -20,15 +20,16 @@ and late arrivals. The earlier resource, logical object-operation, receipt, and readback files remain required. The new `--workload capacity` selector fixes the fleet at three owners and -12 writable Cells. It schedules 10-second points at 2, 4, 16, 64, 256, and -1024 actions per node per second, stopping each shape at its first overloaded +12 writable Cells. The initial CI run scheduled 10-second points at 2, 4, 16, +64, 256, and 1024 actions per node per second. The next run adds 24, 32, 48, +96, 128, and 192 to narrow the overload interval. Each shape stops at its first overloaded point. A point is fully served only when every scheduled action succeeds and admitted work drains within 12 seconds of its 10-second arrival window. The verifier requires at least one fully served point and one overloaded point for uniform, hot, and skewed traffic; it rejects missing arrivals, misstated success, incomplete readback, or an incomplete rate ramp. It reports the last fully served logical write rate separately from the first overloaded -rate. This profile has not yet been run in the isolated provider environment. +rate. The first isolated object-proof result is summarized below. The dedicated `Cell write capacity qualification` GitHub Actions workflow builds one release binary and runs three object-proof repeats on a fresh @@ -37,6 +38,45 @@ and evidence directory. It uploads raw samples, logs, the binary digest, and provider details even if a repeat fails. Its output requires review before a capacity claim; the follower-enabled comparison remains separate. +## First isolated object-proof result + +[CI run 36649534205](https://github.com/crabbuild/cellule/actions/runs/36649534205) +passed three fresh-provider repeats on the same binary, with 12 Cells and +readback verification in every run. The source snapshot was +`4837fc823c198f51a85b01528a6c7aaf6b1f4470`; the release integration +binary SHA-256 was +`2b3aa0b36df5d2c278fe2f3e524a6258c5f886ddedab927057bd4c8db76834a3`. +The RustFS image was pinned at +`sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. +All response proofs were Object; each verified window reported zero root +sequence lag at its end. + +| Shape | Highest fully served offered rate | Fully served logical writes/s | First overloaded offered rate | Fully served action p95 / p99 across repeats | +| --- | ---: | ---: | ---: | ---: | +| Uniform writes | 16 actions/node/s | 47.99–48.00 | 64 actions/node/s | 17.68–32.05 / 120.89–161.14 ms | +| One hot writable Cell | 16 actions/node/s | 47.99–48.00 | 64 actions/node/s | 40.84–52.06 / 56.45–76.39 ms | +| Skewed, 20% writes | 64 actions/node/s | 38.39–38.40 | 256 actions/node/s | 18.26–20.21 / 36.98–43.24 ms | + +These are tested lower bounds, not precise saturation points. The overloaded +uniform windows had client concurrency rejection and owner capacity refusals; +hot windows had 976–982 owner capacity refusals plus 47–55 writes whose +receipt-bound read failed; skewed windows had 1,911–2,227 scheduler-late +arrivals and 469–667 read failures. The upper rate cannot be reported as +sustainable even though some writes completed. No node showed cgroup CPU +throttling. Capture p95 stayed below 2 ms in overloaded windows, whereas +root preparation and response waits were much longer. The current aggregate +provider counters cannot isolate GET/HEAD/PUT time inside preparation or +separate owner admission queueing from provider wait. A second run with raw +per-operation provider timing is needed before choosing one limiter. + +The three `verification.json` files have SHA-256 digests +`35b859c9f9f66aa46da1c797fae5d691a52af907760cf3c4bacc7d79ec809fe2`, +`8a3b951a95a1b9e2b2b00ab8954df53c2783acf352cf529f684fc00be2cd6111`, +and `9d13723e28d4989a87dc60d88b7c84780c125d716ef1d3b1504eb1895fd9efc7` +for repeats 1–3 respectively. Each report contains hashes of its raw TSVs. +The downloaded artifact is under +`$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36649534205`. + After preparing a fresh source/binary snapshot with the Compose qualification guide, run each repeat with a new state directory and Compose project: diff --git a/crates/cellule-app/qualification/entities.py b/crates/cellule-app/qualification/entities.py index 15ab2ec..c4f48fb 100644 --- a/crates/cellule-app/qualification/entities.py +++ b/crates/cellule-app/qualification/entities.py @@ -3,6 +3,7 @@ from __future__ import annotations import bisect +from collections import Counter import csv import hashlib from pathlib import Path @@ -10,7 +11,9 @@ STAGES = (3, 5, 10, 20) SHAPES = ("uniform", "hot", "skewed") POINTS = ((1, 4), (4, 16), (16, 64)) -CAPACITY_POINTS = ((2, 8), (4, 16), (16, 64), (64, 128), (256, 256), (1024, 256)) +CAPACITY_POINTS = ((2, 8), (4, 16), (16, 64), (24, 96), (32, 128), + (48, 192), (64, 256), (96, 256), (128, 256), + (192, 256), (256, 256), (1024, 256)) CELLS_PER_NODE = 4 SECONDS = 10 CAPACITY_DRAIN_GRACE_US = 2_000_000 @@ -40,6 +43,17 @@ def distribution(values: list[int]) -> dict: }) +def verify_object_operations(control: Path, node: int, observations: list[dict]) -> list[dict]: + samples = rows(control / f"node-{node}-object-operations.tsv") + assert Counter((row["operation"], row["outcome"]) for row in samples) == { + (row["operation"], row["outcome"]): int(row["count"]) for row in observations + if int(row["count"]) > 0}, "object operation samples disagree with counters" + for row in samples: + assert int(row["at_ms"]) > 0 + assert all(int(row[key]) >= 0 for key in ("duration_us", "bytes_read", "bytes_written")) + return samples + + def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dict: responses = rows(control / f"node-{node}-responses.tsv") publications = rows(control / f"node-{node}-publications.tsv") @@ -147,7 +161,7 @@ def verify_window(control: Path, nodes: int, shape: str, rate_per_node: int, planned = rate * SECONDS samples = sorted(rows(control / f"{label}.tsv"), key=lambda row: int(row["arrival"])) assert [int(row["arrival"]) for row in samples] == list(range(planned)), "missing or duplicate arrival" - successes, arrival_latencies, scheduled_latencies, writes = [], [], [], [0] * nodes + successes, arrival_latencies, scheduled_latencies, writes, actions = [], [], [], [0] * nodes, [0] * nodes outcomes, intervals = {}, [] new_positions = {entity: [] for entity in range(nodes * CELLS_PER_NODE)} for sample in samples: @@ -180,6 +194,7 @@ def verify_window(control: Path, nodes: int, shape: str, rate_per_node: int, assert read_sequence >= max(sequence, 1) assert sequence > 0 if sample["kind"] == "write" else sequence == 0 successes.append(elapsed) + actions[entity // CELLS_PER_NODE] += 1 arrival_latencies.append(started + elapsed - scheduled) else: assert read_sequence == count == 0 @@ -208,6 +223,7 @@ def verify_window(control: Path, nodes: int, shape: str, rate_per_node: int, return dict(nodes=nodes, shape=shape, rate_per_node=rate_per_node, concurrency=concurrency, planned=planned, outcomes=outcomes, fully_served_arrivals=len(successes) == planned, acknowledged_writes_by_node=writes, completed_actions=len(successes), + completed_actions_by_node=actions, latest_write_sequence_by_entity={entity: max(values, default=0) for entity, values in positions.items()}, completed_writes_per_second=sum(writes) * 1_000_000 / elapsed_us, @@ -292,6 +308,7 @@ def verify_entities(control: Path, capacity: bool = False) -> dict: assert len(observations) == 99 and len({(row["operation"], row["outcome"]) for row in observations}) == 99 assert all(int(row["count"]) >= 0 for row in observations) assert any(row["operation"] == "put" and row["outcome"] == "success" and int(row["count"]) > 0 for row in observations) + object_operations = verify_object_operations(control, node, observations) waits = [int(row["object_wait_us"]) for row in rows(control / f"node-{node}-durability.tsv")] assert waits and min(waits) >= 0 resources[node] = dict(samples=len(samples), max_memory_current_bytes=max(int(row["memory_current_bytes"]) for row in samples), @@ -315,6 +332,19 @@ def verify_entities(control: Path, capacity: bool = False) -> dict: max_unpublished_node_log_bytes=max(int(row["unpublished_node_log_bytes"]) for row in observed), max_disk_file_bytes=max(int(row["disk_bytes"]) for row in observed), max_worker_jobs=max(int(row["worker_jobs"]) for row in observed)) + selected_objects = [row for row in object_operations + if window["started_ms"] <= int(row["at_ms"]) <= window["ended_ms"]] + window.setdefault("node_store", {})[node] = dict( + operations={operation: distribution([int(row["duration_us"]) for row in selected_objects + if row["operation"] == operation]) + for operation in sorted({row["operation"] for row in selected_objects})}, + outcomes=dict(Counter(row["outcome"] for row in selected_objects)), + requests_per_acknowledged_write=(len(selected_objects) + / max(1, window["acknowledged_writes_by_node"][node])), + requests_per_completed_action=(len(selected_objects) + / max(1, window["completed_actions_by_node"][node])), + bytes_read=sum(int(row["bytes_read"]) for row in selected_objects), + bytes_written=sum(int(row["bytes_written"]) for row in selected_objects)) extra = {} if capacity: for window in windows: diff --git a/crates/cellule-app/qualification/test_entities.py b/crates/cellule-app/qualification/test_entities.py index 0fde703..5eaf6f8 100644 --- a/crates/cellule-app/qualification/test_entities.py +++ b/crates/cellule-app/qualification/test_entities.py @@ -6,7 +6,7 @@ import unittest from unittest.mock import patch -from entities import destination, verify_capacity_windows, verify_timing_evidence, verify_window +from entities import destination, verify_capacity_windows, verify_object_operations, verify_timing_evidence, verify_window class EntityWindowEvidence(unittest.TestCase): @@ -197,5 +197,18 @@ def test_rate_mislabeled_as_fully_served_is_rejected(self): verify_capacity_windows(self.root, {}) +class ObjectOperationEvidence(unittest.TestCase): + def test_missing_provider_operation_is_rejected(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) / "node-0-object-operations.tsv" + path.write_text("at_ms\toperation\toutcome\tduration_us\tbytes_read\tbytes_written\n" + "100000\tput\tsuccess\t500\t0\t4096\n") + expected = [dict(operation="put", outcome="success", count="2")] + with self.assertRaisesRegex(AssertionError, "samples disagree"): + verify_object_operations(Path(root), 0, expected) + expected[0]["count"] = "1" + self.assertEqual(len(verify_object_operations(Path(root), 0, expected)), 1) + + if __name__ == "__main__": unittest.main() diff --git a/crates/cellule-app/tests/entities/process/driver.rs b/crates/cellule-app/tests/entities/process/driver.rs index 6ef3b7f..c95b052 100644 --- a/crates/cellule-app/tests/entities/process/driver.rs +++ b/crates/cellule-app/tests/entities/process/driver.rs @@ -146,7 +146,13 @@ async fn run_entity_process(capacity: bool) { (2, 8), (4, 16), (16, 64), - (64, 128), + (24, 96), + (32, 128), + (48, 192), + (64, 256), + (96, 256), + (128, 256), + (192, 256), (256, 256), (1024, 256), ] diff --git a/crates/cellule-app/tests/entities/process/observation.rs b/crates/cellule-app/tests/entities/process/observation.rs index 0bf6d9c..448fe72 100644 --- a/crates/cellule-app/tests/entities/process/observation.rs +++ b/crates/cellule-app/tests/entities/process/observation.rs @@ -6,7 +6,10 @@ use cellule_store::{StorageObservation, StorageObserver, StorageOperation, Stora use std::{ fs::File, io::{BufWriter, Write}, - sync::atomic::{AtomicU64, Ordering}, + sync::{ + Mutex, + atomic::{AtomicU64, Ordering}, + }, }; #[derive(Default)] @@ -16,6 +19,7 @@ pub(super) struct StorageCounters { outcomes: [[AtomicU64; 9]; 11], bytes_read: AtomicU64, bytes_written: AtomicU64, + samples: Mutex>, } impl StorageObserver for StorageCounters { @@ -31,12 +35,14 @@ impl StorageObserver for StorageCounters { self.bytes_written .fetch_add(observation.bytes_written, Ordering::Relaxed); self.finished.fetch_add(1, Ordering::Relaxed); + self.samples.lock().unwrap().push((now_ms(), observation)); } } pub(super) struct NodeObservations { resources: BufWriter, objects: BufWriter, + object_operations: BufWriter, durability: BufWriter, responses: BufWriter, publications: BufWriter, @@ -56,6 +62,7 @@ impl NodeObservations { Self { resources, objects: create("objects"), + object_operations: create("object-operations"), durability: create("durability"), responses: create("responses"), publications: create("publications"), @@ -139,6 +146,23 @@ impl NodeObservations { .unwrap(); } } + writeln!( + self.object_operations, + "at_ms\toperation\toutcome\tduration_us\tbytes_read\tbytes_written" + ) + .unwrap(); + for (at_ms, observation) in storage.samples.lock().unwrap().iter() { + writeln!( + self.object_operations, + "{at_ms}\t{}\t{}\t{}\t{}\t{}", + observation.operation.label(), + observation.outcome.label(), + observation.duration.as_micros(), + observation.bytes_read, + observation.bytes_written + ) + .unwrap(); + } writeln!(self.durability, "object_wait_us").unwrap(); let waits = durability.object_waits(); assert!(!waits.is_empty()); @@ -211,6 +235,7 @@ impl NodeObservations { writeln!(self.follower_appends, "{at_ms}\t{acknowledged}\t{bytes}").unwrap(); } self.objects.flush().unwrap(); + self.object_operations.flush().unwrap(); self.durability.flush().unwrap(); self.responses.flush().unwrap(); self.publications.flush().unwrap(); diff --git a/plans/002-write-throughput-bottleneck.md b/plans/002-write-throughput-bottleneck.md index 77f7d24..a10b49d 100644 --- a/plans/002-write-throughput-bottleneck.md +++ b/plans/002-write-throughput-bottleneck.md @@ -18,10 +18,11 @@ - **Planned at:** `dc387a8`, 2026-09-29 - **Execution update, 2026-09-29:** Separate response, publication, LTX, capture, and upload observations are implemented. A fixed 12-Cell capacity - selector and verifier are implemented and locally tested. A dedicated - GitHub Actions workflow is ready to run three isolated object-proof repeats. - Its execution, a follower-enabled lane, and a measured-limiter follow-up - remain. + selector and verifier are implemented and locally tested. Three isolated + object-proof repeats passed in CI run 36649534205. They established tested + lower bounds but did not identify one limiter; a finer rate ramp and + per-operation provider timing are being qualified. A follower-enabled lane + and a measured-limiter follow-up remain. ## Why this matters diff --git a/plans/README.md b/plans/README.md index c50c23f..22f4db1 100644 --- a/plans/README.md +++ b/plans/README.md @@ -7,7 +7,7 @@ and process evidence outside the checkout. | Order | Plan | Priority | Effort | Status | | --- | --- | --- | --- | --- | | 1 | [Cut forwarded routing work and prove the capacity gain](001-forwarded-routing-and-capacity.md) | P1 | L | DONE: bounded adapter hint and three paired comparisons; product ingress follow-up is external | -| 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | IN PROGRESS: fixed-Cell selector and CI object-proof runner ready; provider evidence and follower lane pending | +| 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | IN PROGRESS: three object-proof repeats verified; finer provider attribution and follower lane pending | Status values: TODO, IN PROGRESS, DONE, BLOCKED (with reason), REJECTED (with reason). Update the row after executing the plan. From a8c6b4bd2a9c09fc039c44a5c987971f9359473b Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 17:49:11 -0700 Subject: [PATCH 03/14] Measure owner actor queue and SQL worker time in capacity runs --- .../performance/2026-09-29-write-capacity.md | 7 +++++++ crates/cellule-app/qualification/entities.py | 12 +++++++++++ .../qualification/test_entities.py | 10 +++++++++ .../tests/entities/process/observation.rs | 17 +++++++++++++++ .../cellule-app/tests/performance_fixture.rs | 17 +++++++++++++++ .../src/cell/actor/requests.rs | 11 +++++++++- crates/cellule-runtime/src/fleet/telemetry.rs | 21 +++++++++++++++++++ 7 files changed, 94 insertions(+), 1 deletion(-) diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index 6656743..1d9e340 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -19,6 +19,13 @@ arrival-latency distribution over every scheduled action, including rejected and late arrivals. The earlier resource, logical object-operation, receipt, and readback files remain required. +A subsequent instrumentation revision adds `node-N-executions.tsv` with actor +queue wait and SQL worker round trip per attempted command. This observation +ends before durability submission and proof. The verifier reports both +distributions per window, which helps distinguish owner admission and worker +occupancy from storage publication without treating either as a durable +response. The first CI result below predates this file. + The new `--workload capacity` selector fixes the fleet at three owners and 12 writable Cells. The initial CI run scheduled 10-second points at 2, 4, 16, 64, 256, and 1024 actions per node per second. The next run adds 24, 32, 48, diff --git a/crates/cellule-app/qualification/entities.py b/crates/cellule-app/qualification/entities.py index c4f48fb..99aedb0 100644 --- a/crates/cellule-app/qualification/entities.py +++ b/crates/cellule-app/qualification/entities.py @@ -56,12 +56,14 @@ def verify_object_operations(control: Path, node: int, observations: list[dict]) def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dict: responses = rows(control / f"node-{node}-responses.tsv") + executions = rows(control / f"node-{node}-executions.tsv") publications = rows(control / f"node-{node}-publications.tsv") phases = rows(control / f"node-{node}-phases.tsv") captures = rows(control / f"node-{node}-captures.tsv") costs = rows(control / f"node-{node}-publication-costs.tsv") appends = rows(control / f"node-{node}-follower-appends.tsv") assert responses, f"node {node}: missing command response evidence" + assert executions, f"node {node}: missing command execution evidence" assert publications, f"node {node}: missing publication evidence" assert phases and captures and costs, f"node {node}: missing LTX or publication phase evidence" sources = {"Recorded", "Fleet", "Object"} @@ -69,6 +71,10 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic assert row["source"] in sources assert int(row["at_ms"]) > 0 assert int(row["response_us"]) >= int(row["confirmation_us"]) >= 0 + for row in executions: + assert int(row["at_ms"]) > 0 + assert int(row["queue_wait_us"]) >= 0 and int(row["worker_round_trip_us"]) >= 0 + assert row["succeeded"] in {"true", "false"} seen = set() for row in publications: key = (row["cell"], int(row["sequence"])) @@ -101,6 +107,7 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic continue start, end = window["started_ms"], window["ended_ms"] selected_responses = [row for row in responses if start <= int(row["at_ms"]) <= end] + selected_executions = [row for row in executions if start <= int(row["at_ms"]) <= end] selected_publications = [row for row in publications if start <= int(row["at_ms"]) <= end] selected_phases = [row for row in phases if start <= int(row["at_ms"]) <= end] selected_captures = [row for row in captures if start <= int(row["at_ms"]) <= end] @@ -112,6 +119,9 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic response_sources=response_sources, response_latency=distribution([int(row["response_us"]) for row in selected_responses]), confirmation_latency=distribution([int(row["confirmation_us"]) for row in selected_responses]), + actor_queue=distribution([int(row["queue_wait_us"]) for row in selected_executions]), + worker_round_trip=distribution([int(row["worker_round_trip_us"]) for row in selected_executions]), + worker_failures=sum(row["succeeded"] == "false" for row in selected_executions), publication_total=distribution([int(row["total_us"]) for row in selected_publications]), publication_queue=distribution([int(row["queue_wait_us"]) for row in selected_publications]), publication_preparation=distribution([int(row["preparation_us"]) for row in selected_publications]), @@ -136,6 +146,8 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic return dict(response_sources={source: sum(row["source"] == source for row in responses) for source in sorted(sources)}, response_latency=distribution([int(row["response_us"]) for row in responses]), + actor_queue=distribution([int(row["queue_wait_us"]) for row in executions]), + worker_round_trip=distribution([int(row["worker_round_trip_us"]) for row in executions]), publication_total=distribution([int(row["total_us"]) for row in publications]), capture_total=distribution([int(row["total_us"]) for row in captures]), uploaded_objects=sum(int(row["objects"]) for row in costs), diff --git a/crates/cellule-app/qualification/test_entities.py b/crates/cellule-app/qualification/test_entities.py index 5eaf6f8..58547af 100644 --- a/crates/cellule-app/qualification/test_entities.py +++ b/crates/cellule-app/qualification/test_entities.py @@ -84,6 +84,9 @@ def setUp(self): (self.root / "node-0-responses.tsv").write_text( "at_ms\tsource\tresponse_us\tconfirmation_us\n" "100001\tFleet\t1200\t900\n100002\tObject\t2000\t1800\n") + (self.root / "node-0-executions.tsv").write_text( + "at_ms\tqueue_wait_us\tworker_round_trip_us\tsucceeded\n" + "100001\t100\t300\ttrue\n100002\t200\t400\ttrue\n") (self.root / "node-0-publications.tsv").write_text( "at_ms\tcell\tsequence\tqueue_wait_us\tpreparation_us\tauthority_us\ttotal_us\tsucceeded\n" "100003\tcell-1\t1\t100\t1000\t200\t1500\ttrue\n") @@ -103,6 +106,7 @@ def test_response_winner_and_later_publication_are_separate(self): self.assertEqual(report["response_sources"], dict(Fleet=1, Object=1, Recorded=0)) self.assertEqual(self.windows[0]["node_durability"][0]["published_roots_per_second"], 0.1) self.assertEqual(self.windows[0]["node_durability"][0]["uploaded_objects"], 2) + self.assertEqual(self.windows[0]["node_durability"][0]["actor_queue"]["count"], 2) def test_incomplete_timing_evidence_is_rejected(self): (self.root / "node-0-publications.tsv").write_text( @@ -110,6 +114,12 @@ def test_incomplete_timing_evidence_is_rejected(self): with self.assertRaisesRegex(AssertionError, "missing publication evidence"): verify_timing_evidence(self.root, 0, self.windows) + def test_missing_execution_evidence_is_rejected(self): + (self.root / "node-0-executions.tsv").write_text( + "at_ms\tqueue_wait_us\tworker_round_trip_us\tsucceeded\n") + with self.assertRaisesRegex(AssertionError, "missing command execution evidence"): + verify_timing_evidence(self.root, 0, self.windows) + def test_duplicate_publication_is_rejected(self): path = self.root / "node-0-publications.tsv" lines = path.read_text().splitlines() diff --git a/crates/cellule-app/tests/entities/process/observation.rs b/crates/cellule-app/tests/entities/process/observation.rs index 448fe72..b84c69c 100644 --- a/crates/cellule-app/tests/entities/process/observation.rs +++ b/crates/cellule-app/tests/entities/process/observation.rs @@ -45,6 +45,7 @@ pub(super) struct NodeObservations { object_operations: BufWriter, durability: BufWriter, responses: BufWriter, + executions: BufWriter, publications: BufWriter, phases: BufWriter, captures: BufWriter, @@ -65,6 +66,7 @@ impl NodeObservations { object_operations: create("object-operations"), durability: create("durability"), responses: create("responses"), + executions: create("executions"), publications: create("publications"), phases: create("phases"), captures: create("captures"), @@ -183,6 +185,20 @@ impl NodeObservations { ) .unwrap(); } + writeln!( + self.executions, + "at_ms\tqueue_wait_us\tworker_round_trip_us\tsucceeded" + ) + .unwrap(); + for (at_ms, queue_wait, worker_round_trip, succeeded) in durability.executions() { + writeln!( + self.executions, + "{at_ms}\t{}\t{}\t{succeeded}", + queue_wait.as_micros(), + worker_round_trip.as_micros() + ) + .unwrap(); + } writeln!(self.publications, "at_ms\tcell\tsequence\tqueue_wait_us\tpreparation_us\tauthority_us\ttotal_us\tsucceeded").unwrap(); for (at_ms, cell, timing) in durability.publications() { writeln!( @@ -238,6 +254,7 @@ impl NodeObservations { self.object_operations.flush().unwrap(); self.durability.flush().unwrap(); self.responses.flush().unwrap(); + self.executions.flush().unwrap(); self.publications.flush().unwrap(); self.phases.flush().unwrap(); self.captures.flush().unwrap(); diff --git a/crates/cellule-app/tests/performance_fixture.rs b/crates/cellule-app/tests/performance_fixture.rs index b893da2..9241790 100644 --- a/crates/cellule-app/tests/performance_fixture.rs +++ b/crates/cellule-app/tests/performance_fixture.rs @@ -148,6 +148,7 @@ pub(super) struct PerfFixture { pub(super) struct DurabilityRecorder { proofs: Mutex>, responses: Mutex>, + executions: Mutex>, publications: Mutex>, phases: Mutex>, captures: Mutex>, @@ -169,6 +170,10 @@ impl DurabilityRecorder { self.responses.lock().unwrap().clone() } + pub(super) fn executions(&self) -> Vec<(i64, Duration, Duration, bool)> { + self.executions.lock().unwrap().clone() + } + pub(super) fn publications(&self) -> Vec<(i64, cellule_runtime::CellId, PublicationTiming)> { self.publications.lock().unwrap().clone() } @@ -207,6 +212,18 @@ impl CellTelemetry for DurabilityRecorder { .push((now_ms(), source, elapsed, confirmation)); } + fn command_execution( + &self, + queue_wait: Duration, + worker_round_trip: Duration, + succeeded: bool, + ) { + self.executions + .lock() + .unwrap() + .push((now_ms(), queue_wait, worker_round_trip, succeeded)); + } + fn publication_completed(&self, cell: cellule_runtime::CellId, timing: PublicationTiming) { self.publications .lock() diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index f9e2bed..7e9fc26 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -146,11 +146,12 @@ pub(super) async fn execute_command( effect_id: u64, ) -> TaskResult { let execution_started = std::time::Instant::now(); + let queue_wait = command.queued_at.elapsed(); tracing::debug!( target: "cellule_runtime::action", parent: &command.trace, event = "cell_execution_started", - actor_queue_us = command.queued_at.elapsed().as_micros(), + actor_queue_us = queue_wait.as_micros(), ); let deadline = SqlDeadline::new(std::time::Instant::now() + SQL_WALL_DEADLINE); let execution = match command.handler.take() { @@ -198,6 +199,11 @@ pub(super) async fn execute_command( match tokio::time::timeout_at(deadline.at().into(), &mut operation).await { Ok(result) => result, Err(_) => { + command.telemetry.command_execution( + queue_wait, + execution_started.elapsed(), + false, + ); let fenced = !deadline.cancel_queued(); tracing::warn!(cell = ?command.cell, sql_started = fenced, "Cell SQL command deadline expired"); if fenced { @@ -227,6 +233,9 @@ pub(super) async fn execute_command( } None => Err(Error::Fenced), }; + command + .telemetry + .command_execution(queue_wait, execution_started.elapsed(), execution.is_ok()); tracing::debug!( target: "cellule_runtime::action", parent: &command.trace, diff --git a/crates/cellule-runtime/src/fleet/telemetry.rs b/crates/cellule-runtime/src/fleet/telemetry.rs index 9ce026a..4541409 100644 --- a/crates/cellule-runtime/src/fleet/telemetry.rs +++ b/crates/cellule-runtime/src/fleet/telemetry.rs @@ -154,6 +154,16 @@ pub trait CellTelemetry: Send + Sync { ) { } + /// Records the actor queue wait and SQL worker round trip for one command. + /// The outcome describes worker execution, before durability proof. + fn command_execution( + &self, + _queue_wait: Duration, + _worker_round_trip: Duration, + _succeeded: bool, + ) { + } + /// Records background root progress separately from the response winner. /// Cell IDs and sequences are for local trace correlation, never metric labels. fn publication_completed(&self, _cell: CellId, _timing: PublicationTiming) {} @@ -258,6 +268,17 @@ impl CellTelemetryHandle { } } + pub(crate) fn command_execution( + &self, + queue_wait: Duration, + worker_round_trip: Duration, + succeeded: bool, + ) { + if let Some(telemetry) = self.inner.get() { + telemetry.command_execution(queue_wait, worker_round_trip, succeeded); + } + } + pub(crate) fn publication_completed(&self, cell: CellId, timing: PublicationTiming) { if let Some(telemetry) = self.inner.get() { telemetry.publication_completed(cell, timing); From 442aed567e08aa2f891e0bd5ec35307f3832bcc6 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 18:03:23 -0700 Subject: [PATCH 04/14] Report finer object-proof capacity evidence --- .../performance/2026-09-29-write-capacity.md | 45 ++++++++++++++++--- 1 file changed, 40 insertions(+), 5 deletions(-) diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index 1d9e340..c0b8098 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -28,15 +28,15 @@ response. The first CI result below predates this file. The new `--workload capacity` selector fixes the fleet at three owners and 12 writable Cells. The initial CI run scheduled 10-second points at 2, 4, 16, -64, 256, and 1024 actions per node per second. The next run adds 24, 32, 48, -96, 128, and 192 to narrow the overload interval. Each shape stops at its first overloaded +64, 256, and 1024 actions per node per second. The second run added 24, 32, +48, 96, 128, and 192 to narrow the overload interval. Each shape stops at its first overloaded point. A point is fully served only when every scheduled action succeeds and admitted work drains within 12 seconds of its 10-second arrival window. The verifier requires at least one fully served point and one overloaded point for uniform, hot, and skewed traffic; it rejects missing arrivals, misstated success, incomplete readback, or an incomplete rate ramp. It reports the last fully served logical write rate separately from the first overloaded -rate. The first isolated object-proof result is summarized below. +rate. The first and second isolated object-proof results are summarized below. The dedicated `Cell write capacity qualification` GitHub Actions workflow builds one release binary and runs three object-proof repeats on a fresh @@ -73,8 +73,8 @@ sustainable even though some writes completed. No node showed cgroup CPU throttling. Capture p95 stayed below 2 ms in overloaded windows, whereas root preparation and response waits were much longer. The current aggregate provider counters cannot isolate GET/HEAD/PUT time inside preparation or -separate owner admission queueing from provider wait. A second run with raw -per-operation provider timing is needed before choosing one limiter. +separate owner admission queueing from provider wait. The second run below +adds raw per-operation provider timing. The three `verification.json` files have SHA-256 digests `35b859c9f9f66aa46da1c797fae5d691a52af907760cf3c4bacc7d79ec809fe2`, @@ -84,6 +84,41 @@ for repeats 1–3 respectively. Each report contains hashes of its raw TSVs. The downloaded artifact is under `$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36649534205`. +## Finer object-proof result + +[CI run 36651191247](https://github.com/crabbuild/cellule/actions/runs/36651191247) +passed three fresh-provider repeats with all acknowledged writes verified by +readback. The source snapshot was +`f46c98c66a78c84bac2244eb739d7548d4ceb056`; the release binary SHA-256 +was `b75352d08f379318f0a872fce7286d13590715eb49aecf7d53cc2cc7906bca7a`. +All response proofs were Object, every window ended with zero root sequence +lag, and no node reported CPU throttling. + +| Shape | Highest fully served offered rate across repeats | Fully served logical writes/s | First overloaded offered rate | +| --- | --- | --- | --- | +| Uniform writes | 24, 32, 32 actions/node/s | 71.96, 95.89, 95.88 | 32, 48, 48 actions/node/s | +| One hot writable Cell | 24, 24, 16 actions/node/s | 71.99, 71.92, 47.99 | 32, 32, 24 actions/node/s | +| Skewed, 20% writes | 24, 24, 16 actions/node/s | 14.40, 14.40, 9.60 | 32, 32, 24 actions/node/s | + +The threshold varies between repeats. Uniform overload mixed scheduler-late +arrivals with owner refusals in two repeats; hot overload mainly returned +owner `not_started` capacity refusals. Skewed overload was only 3–5 +scheduler-late arrivals despite low CPU use, so it is a harness scheduling +limit, not evidence of a Cell write limit. At hot overload, owner 0 response +p95 was 77–169 ms while publication p95 was 36–42 ms and capture p95 stayed +under 1 ms. Its provider PUT p95 was 6.7–7.5 ms, GET p95 1.9–2.1 ms, and +HEAD p95 1.4–1.6 ms. Provider operations can overlap, so these percentiles +cannot be added to infer one command's critical path. The gap between owner +response and publication requires the actor queue and SQL worker timings added +after this run before choosing a write-path optimization. + +The three `verification.json` SHA-256 digests are +`1624b007bd18b19b0e5b498a4cdc307fcfdc30a2320c2ebbd0271f097838c31f`, +`11d03fdc4835bd00a9fc0c90211cb9fd3a21d543ef4f91dc9abb4a64dd15d51c`, +and `d0e41bf722a143149742c47477885d7ea4442a7717d09f7cb6083c916df03e43`. +The downloaded artifact is under +`$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36651191247`. + After preparing a fresh source/binary snapshot with the Compose qualification guide, run each repeat with a new state directory and Compose project: From a01b3483e74f7b0ffcfe3a6765f51663fa8d3ffe Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 18:36:54 -0700 Subject: [PATCH 05/14] Plan controlled hot Cell publication optimization --- .../performance/2026-09-29-write-capacity.md | 25 ++++ plans/002-write-throughput-bottleneck.md | 13 +- .../003-hot-cell-publication-critical-path.md | 131 ++++++++++++++++++ plans/README.md | 8 +- 4 files changed, 170 insertions(+), 7 deletions(-) create mode 100644 plans/003-hot-cell-publication-critical-path.md diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index c0b8098..cbf8553 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -119,6 +119,31 @@ and `d0e41bf722a143149742c47477885d7ea4442a7717d09f7cb6083c916df03e43`. The downloaded artifact is under `$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36651191247`. +## Actor and worker timing attempt + +[CI run 36653277555, attempt 1](https://github.com/crabbuild/cellule/actions/runs/36653277555) +passed readback and evidence integrity in all three repeats. The source +snapshot was `7e1c89c20d62b403ebcace3d208381bcfdfd6485`; the binary SHA-256 +was `d8778c143fe99b9ff5ff2a4c619bba8f7e8a3cdd7548d1d3a92d6fd64fe22880`. +All response proofs were Object and root lag was zero at window ends. The +fully served uniform rate varied from 2 to 16 actions/node/s, while the first +overloaded skewed rate ranged from 4 to 16. Some windows stopped after a +single scheduler-late arrival at 4 actions/node/s with 0.8–2.1% node CPU use +and no CPU throttling. Worker, publication, and provider tail timings also +spiked at these low rates. This attempt does not yield a stable saturation +point or one repeatable dominant phase. A same-revision rerun on a fresh CI +runner is required before selecting a write-path change. +Attempt 2 built the same revision but could not start the first repeat: +the pinned `bucket-init` image pull returned `toomanyrequests: Data limit +exceeded` from its public registry. It produced no capacity measurements. + +The attempt-1 `verification.json` SHA-256 digests are +`b1282ef781b4ca9d962fb43dbf2949662e589ce7737f730fd276473689483673`, +`288e0b50d79780bf4540b9ca2d988880a8d494ca47285a151d5f91964fbd257f`, +and `d9abe2dc3ca698dda71e586da118bc8af0e7fdf2b97b982d963ff4946d108e18`. +The downloaded artifact is under +`$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36653277555`. + After preparing a fresh source/binary snapshot with the Compose qualification guide, run each repeat with a new state directory and Compose project: diff --git a/plans/002-write-throughput-bottleneck.md b/plans/002-write-throughput-bottleneck.md index a10b49d..0e8b56e 100644 --- a/plans/002-write-throughput-bottleneck.md +++ b/plans/002-write-throughput-bottleneck.md @@ -19,10 +19,15 @@ - **Execution update, 2026-09-29:** Separate response, publication, LTX, capture, and upload observations are implemented. A fixed 12-Cell capacity selector and verifier are implemented and locally tested. Three isolated - object-proof repeats passed in CI run 36649534205. They established tested - lower bounds but did not identify one limiter; a finer rate ramp and - per-operation provider timing are being qualified. A follower-enabled lane - and a measured-limiter follow-up remain. + object-proof repeats passed in CI runs 36649534205 and 36651191247. The + finer rate ramp and per-operation provider timing narrowed tested bounds, + but the owner response/publication gap and scheduler-late skewed arrivals + leave the limiter unresolved. Actor queue and SQL worker timing passed + integrity checks in CI run 36653277555 attempt 1, but host scheduling + outliers at 4 actions/node/s made its saturation curves inconclusive. A + same-revision rerun built but could not start because the bucket-init image + registry rate-limited the pull. A follower-enabled lane and a + measured-limiter follow-up remain. ## Why this matters diff --git a/plans/003-hot-cell-publication-critical-path.md b/plans/003-hot-cell-publication-critical-path.md new file mode 100644 index 0000000..6b3a689 --- /dev/null +++ b/plans/003-hot-cell-publication-critical-path.md @@ -0,0 +1,131 @@ +# Plan 003: Reduce the hot Cell's object-publication critical path + +> **Executor:** First establish a stable baseline on a runner with spare CPU +> for the driver, three one-CPU nodes, and RustFS. Do not change root format, +> authority fencing, response proof, or retention semantics to hit a rate. +> Stop if host scheduling or registry failures prevent a clean A/A baseline. + +## Status and evidence + +- **Priority:** P1 for object-proof throughput; follower-proof latency is a + separate lane. +- **Effort:** M for critical-path attribution; implementation effort depends + on the measured subphase. +- **Risk:** LOW for finite timing observations, HIGH for publication changes. +- **Depends on:** [Plan 002](002-write-throughput-bottleneck.md). +- **Status:** TODO. The limiter is narrowed to serial publication, but the + current shared-runner capacity thresholds are inconclusive. + +[The capacity report](../crates/cellule-app/performance/2026-09-29-write-capacity.md) +retains three raw object-proof repeats in each of two CI runs. In the finer +run, the hot owner's first overloaded window published about 61–66 roots/s; +mean publication time was 14–16 ms/root. Owner capacity refusals began at +24–32 actions/node/s, with zero end-window root lag and no node CPU +throttling. Provider PUT p95 was 6.7–7.5 ms, while capture p95 was under +1 ms. A later instrumented run observed 59–67 ms actor-queue p95 in two +hot overload windows while worker round-trip p95 was about 4 ms. Its +fully-served thresholds varied sharply, including scheduler-late arrivals at +only 4 actions/node/s. This supports examining the serialized publication +token; it does **not** establish a safe code change or a stable maximum rate. + +## Scope and invariants + +Inspect `crates/cellule-runtime/src/cell/actor/requests.rs` and +`src/publication/mod.rs` for publication serialization and authority CAS; +`crates/cellule-ltx/src/replica/prepare.rs`, `upload.rs`, and +`directory/mod.rs` for predecessor verification, directory update, and +immutable uploads. The embedding application owns any HTTP, credentials, +follower transport, or deployment changes. + +Preserve one fenced writer; ordered committed outcomes; a successful response +only after an exact authority-pinned root or recoverable follower proof; all +required predecessor and chunk verification; and bounded retention/admission. +An optimization may not skip origin verification merely because local memory +contains a root or digest. Do not alter persisted IDs, object paths, LTX +formats, or signed peer messages. Read each crate's `AGENTS.md`, producers, +consumers, sibling implementations, and tests before editing. + +## Steps + +### 1. Establish a clean baseline + +Run the existing `--workload capacity` hot and uniform shapes at the same +revision and binary digest for three repeats on a dedicated runner with at +least two CPUs beyond the Compose limits for three nodes and the driver. +Record host CPU pressure and runnable-task delay as well as the existing +node cgroups, provider timings, response proof, root drain, readback, and +scheduled arrivals. Pre-pull pinned images once for the run or use a registry +with sufficient quota; retain exact image digests. No rate is fully served +if any arrival is late, refused, or missing. + +**Gate:** All three repeats reach the same fully served and first overloaded +rate interval for the hot shape, with no low-rate scheduler-late arrivals. +If this fails, fix the runner or harness and repeat; do not tune publication. + +### 2. Attribute one root's serial time + +Add finite, nonblocking timing observations around predecessor graph load, +directory update, immutable dependency upload, root-document upload, worker +`bind_prepared`, authority CAS, and worker `confirm_published`. Keep the +observed root sequence in raw local evidence for correlation, never a metric +label. Report overlap explicitly: do not sum parallel upload percentiles. +Measure count and latency of GET/HEAD/range/PUT per acknowledged write and +identify compaction windows separately from ordinary appends. + +**Gate:** For each of three repeats, the same subphase accounts for the +largest avoidable part of the hot Cell's serial critical path. Confirm that +queue growth begins only when offered writes exceed published roots/s. If +the dominant subphase differs by repeat, report the split and stop. + +### 3. Change only the measured subphase + +- If verified predecessor reads dominate, reuse only work that can be tied to + the exact current published root and still fail closed on missing origin + metadata. Test restart, takeover, missing predecessor, and retention races. +- If redundant immutable uploads dominate, prove which content-addressed + objects are already required by the current authority-pinned root before + skipping an upload. Invalidate any memo on failure or owner change, and + verify recovery after deleting a required object. +- If directory computation or compaction dominates, reduce that work without + changing the canonical checksum or compaction debt bound. Test byte-identical + roots and restoration across checkpoint and truncate/regrow cases. +- If authority CAS dominates, stop and write a separate authority protocol + proposal; do not omit or defer the fenced CAS. + +Keep one implementation path and no speculative tuning flags. A failed +preparation must leave only unreachable objects; a failed CAS must fence the +owner and cannot release a success response. + +### 4. Prove the change + +Run paired baseline/candidate builds on the same controlled runner, provider +image, Cell count, node limits, and rate schedule, with three alternating +repeats per build. Require at least **20% more fully served hot logical +writes/s** or **20% lower hot p95 response latency at the same offered rate**. +Require p99 to improve or stay within 5% of baseline, zero failed proofs, +zero readback failures, no higher root lag, and no increase in object requests +per logical write. Report published roots/s separately from response rate; +do not count follower-first responses as proof of publisher capacity. + +Run focused runtime/LTX tests, both application integration and parser tests, +format, all-feature check, Clippy, boundaries, module layout, documentation +gates, and the SQL/peer contract validator. Run broad and provider suites in +CI or an isolated verification snapshot using a checkout-specific target +under `$HOME/Workspace/crabbuild-target`. + +## Done criteria + +- [ ] A stable, integrity-verified A/A baseline identifies the same hot + publication subphase in three repeats. +- [ ] One change to that subphase preserves all proof and recovery invariants. +- [ ] Paired evidence meets the numeric rate or latency gate without increased + retry, root lag, failed proof, or object-request pressure. +- [ ] Raw samples, binary/image digests, and checksums are retained outside + the checkout; the dated report and runnable example are updated. + +## STOP conditions + +- A scheduler-late or provider-registry error prevents the baseline. +- A cache would hide a missing required object or a stale authority root. +- A proposed shortcut changes the root format, fenced CAS, or response proof. +- Any acknowledged receipt cannot be reconstructed exactly after owner loss. diff --git a/plans/README.md b/plans/README.md index 22f4db1..0ef0c1c 100644 --- a/plans/README.md +++ b/plans/README.md @@ -7,7 +7,8 @@ and process evidence outside the checkout. | Order | Plan | Priority | Effort | Status | | --- | --- | --- | --- | --- | | 1 | [Cut forwarded routing work and prove the capacity gain](001-forwarded-routing-and-capacity.md) | P1 | L | DONE: bounded adapter hint and three paired comparisons; product ingress follow-up is external | -| 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | IN PROGRESS: three object-proof repeats verified; finer provider attribution and follower lane pending | +| 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | IN PROGRESS: three object-proof measurement rounds; hosted-runner saturation remains inconclusive, follower lane pending | +| 3 | [Reduce the hot Cell publication critical path](003-hot-cell-publication-critical-path.md) | P1 | M+ | TODO: controlled baseline and subphase attribution required before a code change | Status values: TODO, IN PROGRESS, DONE, BLOCKED (with reason), REJECTED (with reason). Update the row after executing the plan. @@ -18,5 +19,6 @@ Plan 001 begins with measurement and keeps an explicit stop gate before adding an owner hint. It covers the optional Cellule peer HTTP adapter. The embedding application owns ingress routing; its integration work requires its own repo. Plan 002 is independent of Plan 001 and measures the hot-Cell and fleet-wide -write limit. Its output is a separate code-change plan for the measured -limiter; it does not guess a new durability policy in advance. +write limit. Plan 003 narrows the next target to serial object publication +and requires a controlled baseline and exact subphase attribution before +changing code or durability policy. From f9b918a9433865b685d53fed0571029971798c47 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 18:57:50 -0700 Subject: [PATCH 06/14] Record measured hot Cell publication ceiling --- .../performance/2026-09-29-write-capacity.md | 37 +++++++++++++++++++ plans/002-write-throughput-bottleneck.md | 8 +++- .../003-hot-cell-publication-critical-path.md | 14 +++++-- plans/README.md | 4 +- 4 files changed, 55 insertions(+), 8 deletions(-) diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index cbf8553..ccecf34 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -144,6 +144,43 @@ and `d9abe2dc3ca698dda71e586da118bc8af0e7fdf2b97b982d963ff4946d108e18`. The downloaded artifact is under `$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36653277555`. +## Final hot Cell attribution + +[CI run 36655966439](https://github.com/crabbuild/cellule/actions/runs/36655966439) +passed three fresh-provider repeats on source snapshot +`3d01ab9494f76cdf7cb249098ce1814e89f2b203` and binary SHA-256 +`fe847a82c53e456b1f71c32d9309eeb6a6e8e864372a183230db9bedf86eb842`. +All acknowledged writes passed readback, all response proofs were Object, +every window ended at zero root lag, and no node reported CPU throttling. + +| Shape | Fully served offered rate across repeats | Logical writes/s | First overloaded rate | +| --- | --- | --- | --- | +| One hot writable Cell | 24, 24, 24 actions/node/s | 71.96–71.99 | 32, 32, 32 actions/node/s | +| Uniform writes | 24, 32, 32 actions/node/s | 71.96–95.92 | 32, 48, 48 actions/node/s | +| Skewed, 20% writes | 96, 64, 96 actions/node/s | 38.40–57.59 | 128, 96, 128 actions/node/s | + +At hot overload, owner 0 published 62.28–65.10 roots/s. Mean publication +time was 15.10–16.00 ms/root, with 11.64–11.69 provider requests per +acknowledged write. Actor queue p95 climbed from 49–83 ms at the fully served +point to 142–160 ms at overload, while SQL worker round-trip p95 was only +2.03–2.28 ms at overload. Publication p95 was 37–43 ms, of which root +preparation p95 was 33–37 ms. Supported hot action p95/p99 varied from +112–198/175–222 ms; overloaded action p95/p99 was 288–322/324–435 ms. +Owner `not_started` capacity refusals began at the 32 actions/node/s point. +Together these observations identify serialized object publication as the +hot Cell throughput limiter on this object-proof profile. They do not yet +distinguish predecessor verification, directory work, immutable uploads, or +compaction within preparation; [Plan 003](../../../plans/003-hot-cell-publication-critical-path.md) +requires that split before code changes. Uniform and skewed thresholds varied +between repeats, so this result is specific to the hot shape. + +The three `verification.json` SHA-256 digests are +`40e4908a169013a80fe873f4aaf0d6f355872545a72212addb0d712281724e31`, +`58740dbf1d5a57ed16b2138c011e2f694a4900b728967dee019b32a5eb187717`, +and `7d379162a637ba97dd197a8bc85c9d3abebe584454f9d7f65a683c065fb289be`. +The downloaded artifact is under +`$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36655966439`. + After preparing a fresh source/binary snapshot with the Compose qualification guide, run each repeat with a new state directory and Compose project: diff --git a/plans/002-write-throughput-bottleneck.md b/plans/002-write-throughput-bottleneck.md index 0e8b56e..d15e726 100644 --- a/plans/002-write-throughput-bottleneck.md +++ b/plans/002-write-throughput-bottleneck.md @@ -26,8 +26,12 @@ integrity checks in CI run 36653277555 attempt 1, but host scheduling outliers at 4 actions/node/s made its saturation curves inconclusive. A same-revision rerun built but could not start because the bucket-init image - registry rate-limited the pull. A follower-enabled lane and a - measured-limiter follow-up remain. + registry rate-limited the pull. Final CI run 36655966439 passed three + repeats with the same hot threshold (24 fully served, 32 overloaded actions + per node per second) and attributed hot backlog to serial object + publication ahead of a roughly 2 ms SQL worker. Plan 003 now targets that + measured limiter. Uniform and skewed thresholds and the follower-enabled + lane remain unresolved. ## Why this matters diff --git a/plans/003-hot-cell-publication-critical-path.md b/plans/003-hot-cell-publication-critical-path.md index 6b3a689..99959b6 100644 --- a/plans/003-hot-cell-publication-critical-path.md +++ b/plans/003-hot-cell-publication-critical-path.md @@ -13,8 +13,8 @@ on the measured subphase. - **Risk:** LOW for finite timing observations, HIGH for publication changes. - **Depends on:** [Plan 002](002-write-throughput-bottleneck.md). -- **Status:** TODO. The limiter is narrowed to serial publication, but the - current shared-runner capacity thresholds are inconclusive. +- **Status:** TODO. Three repeats identify serial object publication as the + hot Cell limiter; the dominant root-preparation subphase remains unmeasured. [The capacity report](../crates/cellule-app/performance/2026-09-29-write-capacity.md) retains three raw object-proof repeats in each of two CI runs. In the finer @@ -25,8 +25,14 @@ throttling. Provider PUT p95 was 6.7–7.5 ms, while capture p95 was under 1 ms. A later instrumented run observed 59–67 ms actor-queue p95 in two hot overload windows while worker round-trip p95 was about 4 ms. Its fully-served thresholds varied sharply, including scheduler-late arrivals at -only 4 actions/node/s. This supports examining the serialized publication -token; it does **not** establish a safe code change or a stable maximum rate. +only 4 actions/node/s. The final CI run 36655966439 found the same hot +threshold in all three repeats: 24 actions/node/s fully served and 32 +overloaded. At overload, owner 0 published 62–65 roots/s at 15–16 ms mean +publication time/root. Actor queue p95 rose to 142–160 ms while SQL worker +p95 stayed near 2 ms. Root preparation p95 was 33–37 ms of 37–43 ms +publication p95. The serialized publication path is the measured hot +throughput limiter; the exact subphase to change is not yet established. +Uniform and skewed thresholds still varied between repeats. ## Scope and invariants diff --git a/plans/README.md b/plans/README.md index 0ef0c1c..3d75900 100644 --- a/plans/README.md +++ b/plans/README.md @@ -7,8 +7,8 @@ and process evidence outside the checkout. | Order | Plan | Priority | Effort | Status | | --- | --- | --- | --- | --- | | 1 | [Cut forwarded routing work and prove the capacity gain](001-forwarded-routing-and-capacity.md) | P1 | L | DONE: bounded adapter hint and three paired comparisons; product ingress follow-up is external | -| 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | IN PROGRESS: three object-proof measurement rounds; hosted-runner saturation remains inconclusive, follower lane pending | -| 3 | [Reduce the hot Cell publication critical path](003-hot-cell-publication-critical-path.md) | P1 | M+ | TODO: controlled baseline and subphase attribution required before a code change | +| 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | IN PROGRESS: hot object-proof limiter measured; uniform/skewed stability and follower lane pending | +| 3 | [Reduce the hot Cell publication critical path](003-hot-cell-publication-critical-path.md) | P1 | M+ | TODO: serial publication measured; controlled baseline and subphase attribution required before a code change | Status values: TODO, IN PROGRESS, DONE, BLOCKED (with reason), REJECTED (with reason). Update the row after executing the plan. From 4b9e10ed6f4e45e17c4919e13cea7e517e712884 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 19:17:40 -0700 Subject: [PATCH 07/14] Stabilize qualification setup and track follower capacity lane --- .../performance/2026-09-29-write-capacity.md | 10 ++ crates/cellule-app/qualification/compose.yaml | 2 +- .../cellule-ltx/tests/cell/roots/directory.rs | 4 + plans/002-write-throughput-bottleneck.md | 15 ++- plans/004-follower-enabled-capacity-lane.md | 111 ++++++++++++++++++ plans/README.md | 3 + 6 files changed, 138 insertions(+), 7 deletions(-) create mode 100644 plans/004-follower-enabled-capacity-lane.md diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index ccecf34..935d4f3 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -136,6 +136,16 @@ runner is required before selecting a write-path change. Attempt 2 built the same revision but could not start the first repeat: the pinned `bucket-init` image pull returned `toomanyrequests: Data limit exceeded` from its public registry. It produced no capacity measurements. +The docs-only rerun on source `f9b918a` ([CI run +36657593245](https://github.com/crabbuild/cellule/actions/runs/36657593245), +attempts 1 and 2) failed at the same image pull before traffic. The +qualification fixture now pins the same AWS CLI 2.27.41 version from Docker +Hub at manifest digest +`sha256:bc6b7bba44ce38f9604ede49c584824af919047ea03fbcc7c7610671fdef95d8`; +the object-store image, resource limits, rate schedule, and verifier are +unchanged. The replacement still needs CI validation. +Its bucket-init completed against a fresh local RustFS Compose volume; the +three-repeat CI workload has not yet been rerun with it. The attempt-1 `verification.json` SHA-256 digests are `b1282ef781b4ca9d962fb43dbf2949662e589ce7737f730fd276473689483673`, diff --git a/crates/cellule-app/qualification/compose.yaml b/crates/cellule-app/qualification/compose.yaml index d596d96..9feb142 100644 --- a/crates/cellule-app/qualification/compose.yaml +++ b/crates/cellule-app/qualification/compose.yaml @@ -72,7 +72,7 @@ services: restart: "no" bucket-init: - image: public.ecr.aws/aws-cli/aws-cli:2.27.41@sha256:1c2d7a51b1ff4f460bc3f1e1ee46a6cef47c7429ad9089ef99012a95e472a1a5 + image: amazon/aws-cli:2.27.41@sha256:bc6b7bba44ce38f9604ede49c584824af919047ea03fbcc7c7610671fdef95d8 environment: *storage command: [--endpoint-url, http://rustfs:9000, s3api, create-bucket, --bucket, cellule-reference-app] depends_on: diff --git a/crates/cellule-ltx/tests/cell/roots/directory.rs b/crates/cellule-ltx/tests/cell/roots/directory.rs index d398ba0..5d9c551 100644 --- a/crates/cellule-ltx/tests/cell/roots/directory.rs +++ b/crates/cellule-ltx/tests/cell/roots/directory.rs @@ -106,6 +106,9 @@ async fn directory_cache_survives_replica_restart_without_directory_origin_read( let incarnation = [86; 16]; let cache_host = Host::default() .with_local_disk_budget(DiskBudget::new(64 * 1024 * 1024)) + // Optional cache fills use try-acquire. Give this fixture its own + // admission slots so parallel tests cannot suppress the warm fill. + .with_job_slots(Arc::new(tokio::sync::Semaphore::new(4))) .with_directory_cache(cache_root.clone()) .await .unwrap(); @@ -146,6 +149,7 @@ async fn directory_cache_survives_replica_restart_without_directory_origin_read( })); let cached_host = Host::default() .with_local_disk_budget(DiskBudget::new(64 * 1024 * 1024)) + .with_job_slots(Arc::new(tokio::sync::Semaphore::new(4))) .with_directory_cache(cache_root) .await .unwrap(); diff --git a/plans/002-write-throughput-bottleneck.md b/plans/002-write-throughput-bottleneck.md index d15e726..3fea036 100644 --- a/plans/002-write-throughput-bottleneck.md +++ b/plans/002-write-throughput-bottleneck.md @@ -31,7 +31,8 @@ per node per second) and attributed hot backlog to serial object publication ahead of a roughly 2 ms SQL worker. Plan 003 now targets that measured limiter. Uniform and skewed thresholds and the follower-enabled - lane remain unresolved. + lane remain unresolved. Plan 004 specifies the required networked follower + fixture and separate capacity evidence. ## Why this matters @@ -191,14 +192,16 @@ measurements and a regression test for its corresponding invariant. Update ## Done criteria -- [ ] Response-winning proof and later publication are separately measured. -- [ ] A new scheduled workload covers hot, uniform, and skewed Cells without +- [x] Response-winning proof and later publication are separately measured. +- [x] A new scheduled workload covers hot, uniform, and skewed Cells without changing the existing mixed-reader qualification contract. -- [ ] Three repeats per shape distinguish fully served from overloaded rates +- [x] Three repeats per shape distinguish fully served from overloaded rates and retain raw readback, resource, and object-store evidence. -- [ ] A single measured bottleneck has a follow-up implementation plan, or +- [x] A single measured bottleneck has a follow-up implementation plan, or the report says precisely why the evidence is inconclusive. -- [ ] Focused tests, parser tests, format, lint, boundaries, and docs checks pass. +- [x] Focused tests, parser tests, format, lint, boundaries, and docs checks pass. +- [ ] The follower-enabled lane runs separately with its own proof, response, + root-drain, and recovery evidence on the same scheduled shapes. ## STOP conditions diff --git a/plans/004-follower-enabled-capacity-lane.md b/plans/004-follower-enabled-capacity-lane.md new file mode 100644 index 0000000..813f914 --- /dev/null +++ b/plans/004-follower-enabled-capacity-lane.md @@ -0,0 +1,111 @@ +# Plan 004: Qualify follower-proof write capacity across three processes + +> **Executor:** This finishes Plan 002's separate follower-enabled lane. Keep +> its object-proof workload, provider, Cell count, rate schedule, and proof +> rules fixed. A same-process `LocalFollowerTransport` is not evidence for +> networked follower capacity. Stop if an acknowledged follower cut cannot be +> recovered after owner loss. + +## Status and boundary + +- **Priority:** P1 for response-latency comparison. +- **Effort:** L; the application fixture currently has no networked + `NodeLogTransport` or authority enrollment adapter. +- **Risk:** HIGH for transport authorization, CAS ordering, and recovery. +- **Depends on:** [Plan 002](002-write-throughput-bottleneck.md). +- **Status:** TODO. + +The existing three-process fixture in +`crates/cellule-app/tests/entities/process.rs` starts object-only nodes. +`crates/cellule-app/tests/process_node.rs` advertises nodes and renews leases, +but does not install follower stores or a durability provider. Runtime +`node/log_transport.rs` defines the transport contract and a deliberately +in-process test implementation; `node/directory/log.rs` owns authoritative +enrollment, activation, coverage, rotation, and recovery authorization. +Keep the network protocol and credentials in the embedding application or its +test fixture, outside `cellule-runtime` and `cellule-app` production APIs. + +## Steps + +### 1. Add a private-disk, authenticated follower endpoint + +Extend the process fixture with a bounded peer listener for append, seal, +retire, and paged tail. Give each node a `FollowerStore` on its existing +private `/scratch` volume. Authenticate the caller and exact session, member, +epoch, operation, and request deadline before touching that store. Use +`NodeDirectory::authorize_log_append`, `authorize_log_retire`, and +`authorize_log_recovery` as appropriate; make frame and response limits +explicit. The client implements `NodeLogTransport` and connects to the +advertised peer endpoint. Exercise refusal of wrong member, stale epoch, +unauthorized retire, oversized batch, and expired deadline in focused fixture +tests. Never treat an HTTP/TCP acknowledgment as follower fsync unless the +store returned its authenticated receipt. + +### 2. Enroll one host-owned durability generation + +In `tests/process_node.rs`, install the follower store and a +`NodeDurabilityProvider` before `host.start()`. The provider recruits a +complete ensemble through `NodeDirectory::try_recruit_log`, then builds a +`NodeDurabilityConfig` for the exact host session, node, epoch, members, +transport, lease, limits, and telemetry. Serialize its activation, coverage, +rotation, withdrawal, and heartbeat refresh against the same versioned +advertisement; reconcile ambiguous CAS only when the exact session and epoch +still match. Keep the host task group responsible for shutdown and drain. + +Add a readiness marker only after all three nodes are live, follower stores +are listening, and each owner has an enrolled generation. Verify the first +follower-proof response is preceded by all-member fsync and authoritative +activation. Test clean rotation, rejected follower append, owner loss before +object publication, sealed-tail replay, and byte-identical root recovery. + +### 3. Run the same scheduled shapes in an isolated lane + +Add a `capacity-follower` selector to +`crates/cellule-app/tests/entities/process/driver.rs` and +`qualification/scale.py`. Reuse the 12-Cell hot, uniform, and skewed schedule, +arrival accounting, minimum-receipt readback, provider counters, and resource +sampling from the object-proof lane. Record response winner, follower append +bytes and latency, node-log epoch and covered sequence, publication drain, +root lag, retries, and p50/p95/p99 for successful and all scheduled actions. +The parser rejects missing proof records, false fully-served labels, +unrecovered acknowledged writes, or publisher backlog at the end of a drain +window. Keep the existing 300-write qualification profile unchanged. + +Add a separate CI job in `.github/workflows/write-capacity.yml` with three +fresh repeats, pinned source/binary/provider digests, fixed resource limits, +and artifact retention. Pre-pull pinned images for the run and retain their +digests so registry quota cannot be mistaken for workload saturation. Compare +the two lanes only at equal offered rates and report sustainable response +rate and published roots/s separately. Run a cold activation and a steady +resident window in each repeat. + +### 4. Publish decision evidence + +Update `crates/cellule-app/performance/2026-09-29-write-capacity.md` with +per-shape bounds, response-source mix, p95/p99, root drain, provider requests +per logical write, recovery outcome, and checksums of raw external evidence. +Classify each overload using the phase and resource record. If the follower +lane releases responses faster but roots cannot drain, report the stable +response bound separately from publisher capacity. Do not change runtime +durability or publication semantics as part of this measurement. + +## Verification and done criteria + +- [ ] Focused transport authorization, enrollment/CAS, cancellation, and + owner-loss recovery tests pass. +- [ ] Parser tests reject missing follower proof, readback, and root-drain + evidence; format, Clippy, boundaries, module layout, and docs checks pass. +- [ ] Three follower-enabled repeats for each shape have a fully served and + overloaded point with the same fixed schedule and resource profile as + the object-proof lane. +- [ ] Every acknowledged receipt survives owner loss or has an exact + authority-pinned object root; no unsafe replay or false success occurs. +- [ ] Raw logs, samples, source/binary/image digests, and checksums are kept + outside the checkout; the dated report distinguishes response throughput + from eventual publication throughput. + +Run process and provider checks in CI or an isolated snapshot, with +`CARGO_TARGET_DIR` under `$HOME/Workspace/crabbuild-target` for each checkout. +The STOP conditions are any unauthorized follower operation, a successful +response without its durable proof, a failed exact recovery, or a changed +object-proof qualification profile. diff --git a/plans/README.md b/plans/README.md index 3d75900..3f507f7 100644 --- a/plans/README.md +++ b/plans/README.md @@ -9,6 +9,7 @@ and process evidence outside the checkout. | 1 | [Cut forwarded routing work and prove the capacity gain](001-forwarded-routing-and-capacity.md) | P1 | L | DONE: bounded adapter hint and three paired comparisons; product ingress follow-up is external | | 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | IN PROGRESS: hot object-proof limiter measured; uniform/skewed stability and follower lane pending | | 3 | [Reduce the hot Cell publication critical path](003-hot-cell-publication-critical-path.md) | P1 | M+ | TODO: serial publication measured; controlled baseline and subphase attribution required before a code change | +| 4 | [Qualify follower-proof capacity across three processes](004-follower-enabled-capacity-lane.md) | P1 | L | TODO: networked follower transport and authority enrollment fixture required | Status values: TODO, IN PROGRESS, DONE, BLOCKED (with reason), REJECTED (with reason). Update the row after executing the plan. @@ -22,3 +23,5 @@ Plan 002 is independent of Plan 001 and measures the hot-Cell and fleet-wide write limit. Plan 003 narrows the next target to serial object publication and requires a controlled baseline and exact subphase attribution before changing code or durability policy. +Plan 004 supplies the separate follower-enabled lane required by Plan 002; +it does not change the object-proof baseline or the hot publication target. From 141164315ebd414b3521c08caa9fada64bb0bc83 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 19:37:02 -0700 Subject: [PATCH 08/14] Record validated capacity rerun and runner variability --- .../performance/2026-09-29-write-capacity.md | 55 +++++++++++++++---- plans/002-write-throughput-bottleneck.md | 6 +- .../003-hot-cell-publication-critical-path.md | 5 ++ 3 files changed, 53 insertions(+), 13 deletions(-) diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index 935d4f3..57c0357 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -143,9 +143,9 @@ qualification fixture now pins the same AWS CLI 2.27.41 version from Docker Hub at manifest digest `sha256:bc6b7bba44ce38f9604ede49c584824af919047ea03fbcc7c7610671fdef95d8`; the object-store image, resource limits, rate schedule, and verifier are -unchanged. The replacement still needs CI validation. +unchanged. Its bucket-init completed against a fresh local RustFS Compose volume; the -three-repeat CI workload has not yet been rerun with it. +three-repeat CI result is reported below. The attempt-1 `verification.json` SHA-256 digests are `b1282ef781b4ca9d962fb43dbf2949662e589ce7737f730fd276473689483673`, @@ -191,6 +191,40 @@ and `7d379162a637ba97dd197a8bc85c9d3abebe584454f9d7f65a683c065fb289be`. The downloaded artifact is under `$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36655966439`. +## Qualification image rerun + +[CI run 36659182959](https://github.com/crabbuild/cellule/actions/runs/36659182959) +passed all three object-proof repeats after changing only the bucket-init +image registry. The PR test-merge source was +`4b52795bf23cd32b72c820269562b6e3e7d8d0fc`; the release integration +binary SHA-256 was +`8b23d9b7b88c082869a9bcab20e19ebbd345c8165f5c1ea5bb38d545d9d4d850`. +All three reports passed integrity and acknowledged-write readback checks, +every response winner was Object, root lag ended at zero, and the hot owner +reported zero CPU throttling. + +| Shape | Fully served, repeats 1–3 | First overloaded, repeats 1–3 | +| --- | --- | --- | +| One hot writable Cell | 16, 16, 16 actions/node/s | 24, 24, 24 actions/node/s | +| Uniform | 24, 32, 32 actions/node/s | 32, 48, 48 actions/node/s | +| Skewed, 20% writes | 48, 48, 48 actions/node/s | 64, 64, 64 actions/node/s | + +At hot overload, owner 0 published 57.84–59.05 roots/s. Its actor queue +p95 was 136–142 ms, SQL worker round-trip p95 2.53–2.75 ms, publication +p95 37–43 ms, and root preparation p95 32–38 ms. Provider requests per +acknowledged write were 11.66–11.73. This independently supports serial +object publication as the hot Cell limiter, while the lower hot rate interval +than run 36655966439 shows that exact throughput is sensitive to the shared +CI runner. Plan 003 requires a controlled A/A baseline before claiming a +code-change gain. + +The `verification.json` SHA-256 digests for repeats 1–3 are +`5ff25de33a3c9edacd677b3e724b0003d28aceea1f0f2cb3e8f9b729e85d9086`, +`31035c59762a5d34eea412a52b412924f9fded01cca9085c05930a5ba347194c`, +and `bfbef7176b533f326d1e536122a814afbf6e2b7d8bbcfe3184574fcbac0b4940`. +The downloaded artifact is under +`$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36659182959`. + After preparing a fresh source/binary snapshot with the Compose qualification guide, run each repeat with a new state directory and Compose project: @@ -212,7 +246,8 @@ overload point identifies the harness admission limit until runtime and provider phase evidence demonstrates a narrower Cell limiter. The existing entity host does not enroll a follower durability lane, so this -profile cannot supply the separate follower-enabled comparison. The added +profile cannot supply the separate follower-enabled comparison. [Plan 004](../../../plans/004-follower-enabled-capacity-lane.md) +specifies the required fixture and evidence. The added publication observation aggregates root preparation and provider I/O; it cannot by itself distinguish predecessor GET/HEAD, immutable PUT, and provider wait inside that phase. Those limits prevent selecting a safe implementation @@ -240,12 +275,8 @@ above. That binary predates the fixed-Cell capacity selector and cannot run it; create a new source snapshot and release binary for the capacity runs. This is a compile result, not an execution result. -Before choosing a write-path optimization, run the workload in a dedicated -provider environment with a new object prefix per repeat. Keep the same -source/binary digest, node limits, Cell count, and arrival schedule across -three repeats; add a separate follower-enabled lane and an offered rate above -the first overload point. Require readback for every acknowledged write and -compare response proof rates with eventual root drain. If root preparation -dominates, collect its GET/HEAD, PUT, provider wait, and CAS split before -proposing a code change. If the three repeats disagree, the result remains -inconclusive. +Plan 003 requires a dedicated provider environment and a repeatable A/A +rate interval before modifying publication. It then splits root preparation +into predecessor reads, directory work, uploads, provider wait, and CAS. +Plan 004 runs the follower-enabled lane separately, with recovery proof and +eventual root drain alongside response throughput. diff --git a/plans/002-write-throughput-bottleneck.md b/plans/002-write-throughput-bottleneck.md index 3fea036..e773a05 100644 --- a/plans/002-write-throughput-bottleneck.md +++ b/plans/002-write-throughput-bottleneck.md @@ -32,7 +32,11 @@ publication ahead of a roughly 2 ms SQL worker. Plan 003 now targets that measured limiter. Uniform and skewed thresholds and the follower-enabled lane remain unresolved. Plan 004 specifies the required networked follower - fixture and separate capacity evidence. + fixture and separate capacity evidence. CI run 36659182959 passed another + three object-proof repeats after the bucket-init image registry change; + hot throughput was fully served at 16 and overloaded at 24 actions/node/s + in each repeat, confirming the limiting phase while showing runner-dependent + absolute capacity. ## Why this matters diff --git a/plans/003-hot-cell-publication-critical-path.md b/plans/003-hot-cell-publication-critical-path.md index 99959b6..218eae2 100644 --- a/plans/003-hot-cell-publication-critical-path.md +++ b/plans/003-hot-cell-publication-critical-path.md @@ -33,6 +33,11 @@ p95 stayed near 2 ms. Root preparation p95 was 33–37 ms of 37–43 ms publication p95. The serialized publication path is the measured hot throughput limiter; the exact subphase to change is not yet established. Uniform and skewed thresholds still varied between repeats. +An independent three-repeat CI run 36659182959 again found serial object +publication ahead of a roughly 3 ms SQL worker at hot overload, but its hot +interval shifted to 16 fully served and 24 overloaded actions/node/s. The +different interval strengthens the requirement for a dedicated, controlled +A/A baseline before applying this plan's numeric improvement gate. ## Scope and invariants From effc15196b6ae9728890cbf103e2eaa6c67d5043 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 19:57:53 -0700 Subject: [PATCH 09/14] Add networked follower capacity qualification lane --- .github/workflows/write-capacity.yml | 64 ++ Cargo.lock | 2 + crates/cellule-app/Cargo.toml | 2 + crates/cellule-app/qualification/entities.py | 20 +- crates/cellule-app/qualification/run.sh | 3 +- crates/cellule-app/qualification/scale.py | 25 +- .../qualification/test_entities.py | 17 +- crates/cellule-app/tests/entities/process.rs | 37 +- .../tests/entities/process/driver.rs | 15 +- crates/cellule-app/tests/integration.rs | 1 + crates/cellule-app/tests/process_follower.rs | 882 ++++++++++++++++++ crates/cellule-app/tests/process_node.rs | 96 +- 12 files changed, 1124 insertions(+), 40 deletions(-) create mode 100644 crates/cellule-app/tests/process_follower.rs diff --git a/.github/workflows/write-capacity.yml b/.github/workflows/write-capacity.yml index 9966e35..1101b03 100644 --- a/.github/workflows/write-capacity.yml +++ b/.github/workflows/write-capacity.yml @@ -81,3 +81,67 @@ jobs: ${{ runner.temp }}/cell-write-capacity-run-*/evidence/ if-no-files-found: error retention-days: 7 + + follower-proof: + runs-on: ubuntu-24.04 + timeout-minutes: 100 + steps: + - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 + with: + persist-credentials: false + - name: Pin follower lane source and release binary + run: | + set -euo pipefail + build_state="$RUNNER_TEMP/cell-follower-capacity-build" + printf 'CELLULE_FOLLOWER_BUILD_STATE=%s\n' "$build_state" >> "$GITHUB_ENV" + mkdir -p "$build_state/source" "$build_state/target-linux" "$build_state/evidence" + git archive HEAD | tar -x -C "$build_state/source" + git rev-parse HEAD > "$build_state/evidence/source-revision.txt" + uname -a > "$build_state/evidence/host.txt" + docker info --format '{{json .}}' > "$build_state/evidence/docker-info.json" + python3 -B -m unittest discover \ + -s crates/cellule-app/qualification -p 'test_*.py' + CELLULE_REFERENCE_STATE="$build_state" \ + docker compose -p cell-follower-capacity-build \ + -f "$build_state/source/crates/cellule-app/qualification/compose.yaml" \ + pull --quiet build rustfs bucket-init + CELLULE_REFERENCE_STATE="$build_state" \ + docker compose -p cell-follower-capacity-build \ + -f "$build_state/source/crates/cellule-app/qualification/compose.yaml" \ + run --name cell-follower-capacity-build build \ + 2>&1 | tee "$build_state/evidence/build.log" + binary=$(find "$build_state/target-linux/release/deps" -maxdepth 1 \ + -type f -executable -name 'integration-*') + test -n "$binary" + test "$(printf '%s\n' "$binary" | wc -l)" -eq 1 + sha256sum "$binary" > "$build_state/evidence/binary.sha256" + - name: Run three fresh follower-proof capacity repeats + run: | + set -euo pipefail + for repeat in 1 2 3; do + run_state="$RUNNER_TEMP/cell-follower-capacity-run-$repeat" + mkdir -p "$run_state/evidence" + ln -s "$CELLULE_FOLLOWER_BUILD_STATE/source" "$run_state/source" + ln -s "$CELLULE_FOLLOWER_BUILD_STATE/target-linux" "$run_state/target-linux" + cp "$CELLULE_FOLLOWER_BUILD_STATE/evidence/source-revision.txt" "$run_state/evidence/" + cp "$CELLULE_FOLLOWER_BUILD_STATE/evidence/host.txt" "$run_state/evidence/" + cp "$CELLULE_FOLLOWER_BUILD_STATE/evidence/docker-info.json" "$run_state/evidence/" + chmod 1777 "$run_state/evidence" + python3 "$run_state/source/crates/cellule-app/qualification/scale.py" \ + --state "$run_state" --project "cell-follower-capacity-r$repeat" \ + --workload capacity-follower + done + - name: Retain provider image identity + if: always() + run: | + docker image ls -a --digests --no-trunc \ + > "$CELLULE_FOLLOWER_BUILD_STATE/evidence/images.txt" + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 + if: always() + with: + name: cell-follower-capacity-${{ github.run_id }}-${{ github.run_attempt }} + path: | + ${{ runner.temp }}/cell-follower-capacity-build/evidence/ + ${{ runner.temp }}/cell-follower-capacity-run-*/evidence/ + if-no-files-found: error + retention-days: 7 diff --git a/Cargo.lock b/Cargo.lock index 1800c73..54de4bf 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -224,6 +224,7 @@ version = "0.1.0" dependencies = [ "async-trait", "blake3", + "bytes", "cellule-host", "cellule-ltx", "cellule-runtime", @@ -231,6 +232,7 @@ dependencies = [ "ed25519-dalek", "futures-util", "object_store", + "prost", "tempfile", "tokio", "tokio-util", diff --git a/crates/cellule-app/Cargo.toml b/crates/cellule-app/Cargo.toml index 962302c..bad0f05 100644 --- a/crates/cellule-app/Cargo.toml +++ b/crates/cellule-app/Cargo.toml @@ -16,6 +16,7 @@ cellule-runtime.workspace = true [dev-dependencies] async-trait.workspace = true +bytes.workspace = true cellule-host.workspace = true cellule-runtime = { workspace = true, features = ["test-support"] } cellule-ltx = { workspace = true, features = ["replica"] } @@ -23,6 +24,7 @@ cellule-store.workspace = true ed25519-dalek.workspace = true futures-util.workspace = true object_store.workspace = true +prost.workspace = true tempfile.workspace = true tokio = { workspace = true, features = ["io-util", "macros", "net", "rt-multi-thread"] } tokio-util.workspace = true diff --git a/crates/cellule-app/qualification/entities.py b/crates/cellule-app/qualification/entities.py index 99aedb0..40f49e5 100644 --- a/crates/cellule-app/qualification/entities.py +++ b/crates/cellule-app/qualification/entities.py @@ -142,7 +142,8 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic uploaded_objects=sum(int(row["objects"]) for row in selected_costs), uploaded_bytes=sum(int(row["bytes"]) for row in selected_costs), follower_appends=len(selected_appends), - follower_append_failures=sum(row["acknowledged"] == "false" for row in selected_appends)) + follower_append_failures=sum(row["acknowledged"] == "false" for row in selected_appends), + follower_append_bytes=sum(int(row["bytes"]) for row in selected_appends)) return dict(response_sources={source: sum(row["source"] == source for row in responses) for source in sorted(sources)}, response_latency=distribution([int(row["response_us"]) for row in responses]), @@ -153,6 +154,8 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic uploaded_objects=sum(int(row["objects"]) for row in costs), uploaded_bytes=sum(int(row["bytes"]) for row in costs), follower_appends=len(appends), + acknowledged_follower_appends=sum(row["acknowledged"] == "true" for row in appends), + follower_append_bytes=sum(int(row["bytes"]) for row in appends), completed_publications=sum(row["succeeded"] == "true" for row in publications), failed_publications=sum(row["succeeded"] == "false" for row in publications)) @@ -271,7 +274,18 @@ def verify_capacity_windows(control: Path, positions: dict[int, list[int]]) -> l return windows -def verify_entities(control: Path, capacity: bool = False) -> dict: +def verify_follower_proof(resources: dict) -> dict: + fleet_proofs = sum(resource["durability"]["response_sources"]["Fleet"] + for resource in resources.values()) + acknowledged_appends = sum(resource["durability"]["acknowledged_follower_appends"] + for resource in resources.values()) + assert fleet_proofs > 0, "follower lane returned no follower-proof responses" + assert acknowledged_appends > 0, "missing acknowledged follower append evidence" + return dict(follower_proof_responses=fleet_proofs, follower_appends=acknowledged_appends) + + +def verify_entities(control: Path, capacity: bool = False, follower: bool = False) -> dict: + assert not follower or capacity stages = (3,) if capacity else STAGES evidence_prefix = "capacity" if capacity else "entity" owners = rows(control / f"{evidence_prefix}-owners.tsv") @@ -359,6 +373,8 @@ def verify_entities(control: Path, capacity: bool = False) -> dict: bytes_written=sum(int(row["bytes_written"]) for row in selected_objects)) extra = {} if capacity: + if follower: + extra.update(verify_follower_proof(resources)) for window in windows: root_lags = {} for entity, acknowledged in window["latest_write_sequence_by_entity"].items(): diff --git a/crates/cellule-app/qualification/run.sh b/crates/cellule-app/qualification/run.sh index 3c45eba..76c58ed 100644 --- a/crates/cellule-app/qualification/run.sh +++ b/crates/cellule-app/qualification/run.sh @@ -9,7 +9,8 @@ case "${1:-}" in entity-node) role="node-${CELLULE_PERF_PROCESS_NODE:?}"; selected=entities::process::entity_process_node ;; entity-scale) role=driver; selected=entities::process::driver::entity_process_scaling ;; entity-capacity) role=driver; selected=entities::process::driver::entity_process_capacity ;; - *) printf 'usage: run.sh node|driver|scale|rollout|entities|entity-node|entity-scale|entity-capacity\n' >&2; exit 2 ;; + entity-capacity-follower) role=driver; selected=entities::process::driver::entity_process_capacity_follower ;; + *) printf 'usage: run.sh node|driver|scale|rollout|entities|entity-node|entity-scale|entity-capacity|entity-capacity-follower\n' >&2; exit 2 ;; esac binary= for candidate in /target/release/deps/integration-*; do diff --git a/crates/cellule-app/qualification/scale.py b/crates/cellule-app/qualification/scale.py index 04d1eae..86dad10 100644 --- a/crates/cellule-app/qualification/scale.py +++ b/crates/cellule-app/qualification/scale.py @@ -126,7 +126,7 @@ def verify_reader_loss(control: Path, killed_node: int) -> dict: class Fleet: def __init__(self, state: Path, project: str, overrides: list[Path], workload: str = "readers"): - assert workload in ("readers", "entities", "capacity") + assert workload in ("readers", "entities", "capacity", "capacity-follower") self.workload = workload self.state = state.resolve(strict=True) self.project = project @@ -136,7 +136,8 @@ def __init__(self, state: Path, project: str, overrides: list[Path], workload: s existing = self.run("docker", "ps", "-aq", "--filter", f"label=com.docker.compose.project={project}") if existing.strip(): raise ValueError("project already has containers; retain it and choose a fresh project") - evidence_name = {"readers": "scaling", "entities": "entity-scaling", "capacity": "entity-capacity"}[workload] + evidence_name = {"readers": "scaling", "entities": "entity-scaling", "capacity": "entity-capacity", + "capacity-follower": "entity-capacity-follower"}[workload] self.evidence = self.state / "evidence" / evidence_name self.evidence.mkdir(mode=0o1777) self.evidence.chmod(0o1777) @@ -151,22 +152,25 @@ def __init__(self, state: Path, project: str, overrides: list[Path], workload: s node = config["services"]["node-0"] # Twenty live nodes plus one killed boot. New boots have new identities; # restarting a deterministic fixture session would bypass expiry fencing. - for index in range(3, 3 if workload == "capacity" else 21): + for index in range(3, 3 if workload.startswith("capacity") else 21): added = copy.deepcopy(node) added["environment"].update( CELLULE_PERF_PROCESS_NODE=str(index), CELLULE_PERF_PROCESS_ADVERTISE=f"node-{index}:8080", ) config["services"][f"node-{index}"] = added - config["services"]["driver"]["command"] = [{"readers": "scale", "entities": "entity-scale", "capacity": "entity-capacity"}[workload]] - if workload in ("entities", "capacity"): + config["services"]["driver"]["command"] = [{"readers": "scale", "entities": "entity-scale", "capacity": "entity-capacity", + "capacity-follower": "entity-capacity-follower"}[workload]] + if workload in ("entities", "capacity", "capacity-follower"): for name, service in config["services"].items(): if name.startswith("node-"): service["command"] = ["entity-node"] - if workload == "capacity": + if workload.startswith("capacity"): for name, service in config["services"].items(): if name.startswith("node-") or name == "driver": service["environment"]["CELLULE_PERF_PROCESS_ROOT"] = f"capacity-{project}" + if workload == "capacity-follower" and name.startswith("node-"): + service["environment"]["CELLULE_PERF_PROCESS_FOLLOWER"] = "1" for service in config["services"].values(): for volume in service.get("volumes", []): if volume["target"] == "/evidence": @@ -275,7 +279,7 @@ def execute(self) -> None: self.verify() def verify(self) -> None: - stages = (3,) if self.workload == "capacity" else (3, 5, 10, 20) + stages = (3,) if self.workload.startswith("capacity") else (3, 5, 10, 20) assert len(self.active) == stages[-1] and len(self.killed) == (1 if self.workload == "readers" else 0) assert [(event["action"], event["argument"]) for event in self.events if event["action"] == "scale"] == [ ("scale", stage) for stage in stages @@ -308,10 +312,11 @@ def verify(self) -> None: binaries.add((self.evidence / f"{role}-binary.sha256").read_text().split()[0]) assert len(binaries) == 1 source = (self.state / "evidence/source-revision.txt").read_text().strip() - if self.workload in ("entities", "capacity"): + if self.workload in ("entities", "capacity", "capacity-follower"): result = dict(workload=self.workload, source=source, binary_sha256=binaries.pop(), roles=reports, events=self.events, - **verify_entities(self.control, capacity=self.workload == "capacity")) + **verify_entities(self.control, capacity=self.workload.startswith("capacity"), + follower=self.workload == "capacity-follower")) (self.evidence / "verification.json").write_text(json.dumps(result, indent=2) + "\n") print(f"Verified entity integrity and resources at {stages} nodes; evidence: {self.evidence}", flush=True) return @@ -345,7 +350,7 @@ def main() -> None: parser.add_argument("--state", type=Path, required=True, help="prepared source, binary and evidence directory") parser.add_argument("--project", required=True, help="fresh Compose project; stopped containers are retained") parser.add_argument("--compose-file", type=Path, action="append", default=[], help="explicit image/cache override") - parser.add_argument("--workload", choices=("readers", "entities", "capacity"), default="readers", help="reader replacement, scaling, or fixed 12-Cell capacity traffic") + parser.add_argument("--workload", choices=("readers", "entities", "capacity", "capacity-follower"), default="readers", help="reader replacement, scaling, or fixed 12-Cell capacity traffic") args = parser.parse_args() fleet = Fleet(args.state, args.project, args.compose_file, args.workload) try: diff --git a/crates/cellule-app/qualification/test_entities.py b/crates/cellule-app/qualification/test_entities.py index 58547af..6dfa3f2 100644 --- a/crates/cellule-app/qualification/test_entities.py +++ b/crates/cellule-app/qualification/test_entities.py @@ -6,7 +6,7 @@ import unittest from unittest.mock import patch -from entities import destination, verify_capacity_windows, verify_object_operations, verify_timing_evidence, verify_window +from entities import destination, verify_capacity_windows, verify_follower_proof, verify_object_operations, verify_timing_evidence, verify_window class EntityWindowEvidence(unittest.TestCase): @@ -220,5 +220,20 @@ def test_missing_provider_operation_is_rejected(self): self.assertEqual(len(verify_object_operations(Path(root), 0, expected)), 1) +class FollowerProofEvidence(unittest.TestCase): + def test_follower_lane_requires_proof_and_acknowledged_append(self): + resources = {0: dict(durability=dict(response_sources=dict(Fleet=0), + acknowledged_follower_appends=1))} + with self.assertRaisesRegex(AssertionError, "no follower-proof responses"): + verify_follower_proof(resources) + resources[0]["durability"]["response_sources"]["Fleet"] = 1 + resources[0]["durability"]["acknowledged_follower_appends"] = 0 + with self.assertRaisesRegex(AssertionError, "missing acknowledged follower append"): + verify_follower_proof(resources) + resources[0]["durability"]["acknowledged_follower_appends"] = 2 + self.assertEqual(verify_follower_proof(resources), + dict(follower_proof_responses=1, follower_appends=2)) + + if __name__ == "__main__": unittest.main() diff --git a/crates/cellule-app/tests/entities/process.rs b/crates/cellule-app/tests/entities/process.rs index 72d8e55..6e9ee74 100644 --- a/crates/cellule-app/tests/entities/process.rs +++ b/crates/cellule-app/tests/entities/process.rs @@ -2,9 +2,11 @@ use super::super::fleet::{GatewayStats, start_gateway_peer_server}; use super::super::performance_fixture::{node_session, now_ms, rustfs_store}; +use super::super::process_follower; use super::super::process_node; use super::super::process_performance::{publish_address, wait_for_marker}; use super::*; +use cellule_runtime::follower::FollowerStore; use cellule_runtime::peer::PeerVerifier; use std::{ env, @@ -61,6 +63,7 @@ async fn entity_process_node() { .parse() .unwrap(); assert!(node < 20); + let follower_enabled = env::var("CELLULE_PERF_PROCESS_FOLLOWER").as_deref() == Ok("1"); let sync = env::var("CELLULE_PERF_PROCESS_SYNC").unwrap(); let sync = Path::new(&sync); let application = compiled_entities(); @@ -76,20 +79,38 @@ async fn entity_process_node() { let listener = TcpListener::bind(env::var("CELLULE_PERF_PROCESS_BIND").unwrap()) .await .unwrap(); + let follower_listener = if follower_enabled { + Some(TcpListener::bind("0.0.0.0:8081").await.unwrap()) + } else { + None + }; let address = tokio::net::lookup_host(env::var("CELLULE_PERF_PROCESS_ADVERTISE").unwrap()) .await .unwrap() .next() .unwrap(); let endpoint = format!("https://{address}"); - let (host, durability, readers) = process_node::start( + let (host, durability, readers) = process_node::start_configured( node, application.clone(), &layout, directory.path(), endpoint.clone(), + follower_enabled, ) .await; + let follower_server = follower_listener.map(|listener| { + let store = host + .owned_component::(cellule_host::FOLLOWER_STORE_COMPONENT) + .unwrap(); + process_follower::serve( + listener, + cellule_runtime::identity::NodeId::from_bytes(*node_session(node).as_bytes()), + (*store).clone(), + process_node::directory(&layout, ®istry), + process_node::signing_key(node), + ) + }); let mut handles = Vec::new(); for entity in node * ENTITIES_PER_NODE..(node + 1) * ENTITIES_PER_NODE { handles.push( @@ -129,6 +150,16 @@ async fn entity_process_node() { )), ); publish_address(&sync.join(format!("node-{node}.ready")), address); + if follower_enabled { + let deadline = Instant::now() + Duration::from_secs(30); + while host.runtime().node_durability().is_none() { + assert!( + Instant::now() < deadline, + "follower enrollment did not complete" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + } publish_marker(&sync.join(format!("node-{node}.serving")), []); let mut observations = observation::NodeObservations::new(sync, node); let mut next_sample = Instant::now(); @@ -174,4 +205,8 @@ async fn entity_process_node() { publish_marker(&sync.join(format!("node-{node}.done")), []); server.abort(); let _ = server.await; + if let Some(server) = follower_server { + server.abort(); + let _ = server.await; + } } diff --git a/crates/cellule-app/tests/entities/process/driver.rs b/crates/cellule-app/tests/entities/process/driver.rs index c95b052..dba700e 100644 --- a/crates/cellule-app/tests/entities/process/driver.rs +++ b/crates/cellule-app/tests/entities/process/driver.rs @@ -50,16 +50,22 @@ struct Sample { #[tokio::test(flavor = "multi_thread", worker_threads = 4)] #[ignore = "Compose controller required for scheduled entity traffic on 3/5/10/20 nodes"] async fn entity_process_scaling() { - run_entity_process(false).await; + run_entity_process(false, false).await; } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] #[ignore = "Compose controller required for fixed-Cell scheduled capacity traffic"] async fn entity_process_capacity() { - run_entity_process(true).await; + run_entity_process(true, false).await; } -async fn run_entity_process(capacity: bool) { +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "Compose controller required for networked follower-proof capacity traffic"] +async fn entity_process_capacity_follower() { + run_entity_process(true, true).await; +} + +async fn run_entity_process(capacity: bool, follower_enabled: bool) { let sync = env::var("CELLULE_PERF_PROCESS_SYNC").unwrap(); let sync = Path::new(&sync); let mut controller = Controller::new(sync); @@ -95,6 +101,9 @@ async fn run_entity_process(capacity: bool) { for node in 0..nodes { wait_for_marker(&sync.join(format!("node-{node}.serving"))).await; } + if follower_enabled { + assert_eq!(nodes, 3); + } let addresses = endpoints(sync, nodes).await; publish_marker(&sync.join("entity-stage.request"), nodes.to_string()); for node in 0..nodes { diff --git a/crates/cellule-app/tests/integration.rs b/crates/cellule-app/tests/integration.rs index a9de1be..97044ee 100644 --- a/crates/cellule-app/tests/integration.rs +++ b/crates/cellule-app/tests/integration.rs @@ -74,6 +74,7 @@ mod host; mod performance; mod performance_fixture; mod primitives; +mod process_follower; mod process_node; mod process_performance; mod process_recruitment; diff --git a/crates/cellule-app/tests/process_follower.rs b/crates/cellule-app/tests/process_follower.rs new file mode 100644 index 0000000..d956e4c --- /dev/null +++ b/crates/cellule-app/tests/process_follower.rs @@ -0,0 +1,882 @@ +//! Test-only authenticated network transport for private-disk follower lanes. + +use super::performance_fixture::now_ms; +use bytes::Bytes; +use cellule_host::{FacilityResult, NodeDurabilityProvider}; +use cellule_ltx::Limits; +use cellule_runtime::fleet::telemetry::CellTelemetryHandle; +use cellule_runtime::follower::{FollowerReceipt, FollowerStore, FollowerTailPage}; +use cellule_runtime::identity::NodeId; +use cellule_runtime::node::durability::{NodeDurabilityConfig, NodeLogAuthority}; +use cellule_runtime::node::lease::NodeLeaseGuard; +use cellule_runtime::node::log::NodeLogRotationBarrier; +use cellule_runtime::node::log_transport::{ + AppendRequest, NodeLogTransport, RetireRequest, SealRequest, TailRequest, +}; +use cellule_runtime::node::{NodeAdvertisement, NodeDirectory, VersionedNodeAdvertisement}; +use cellule_runtime::{Error, Result, SessionId}; +use ed25519_dalek::{Signature, Signer, SigningKey, Verifier}; +use futures_util::future::BoxFuture; +use prost::Message; +use std::{ + collections::HashMap, + future::Future, + net::SocketAddr, + pin::Pin, + sync::{Arc, Mutex as StdMutex}, + time::{Duration, Instant}, +}; +use tokio::{ + io::{AsyncReadExt, AsyncWriteExt}, + net::{TcpListener, TcpStream}, + sync::Mutex, +}; + +const REQUEST_DOMAIN: &[u8] = b"cellule.test-follower.request.v1\0"; +const RESPONSE_DOMAIN: &[u8] = b"cellule.test-follower.response.v1\0"; +const MAX_REQUEST_BYTES: usize = 8 << 20; +const MAX_RESPONSE_BYTES: usize = 2 << 20; +const REQUEST_TIMEOUT: Duration = Duration::from_secs(10); +const MAX_DEADLINE_AHEAD_MS: i64 = 10_000; +const MAX_APPEND_FRAMES: usize = 64; + +#[derive(Clone, PartialEq, Message)] +struct SignedWire { + #[prost(bytes = "vec", tag = "1")] + body: Vec, + #[prost(bytes = "vec", tag = "2")] + signature: Vec, +} + +#[derive(Clone, PartialEq, Message)] +struct RequestWire { + #[prost(bytes = "vec", tag = "1")] + sender: Vec, + #[prost(bytes = "vec", tag = "2")] + member: Vec, + #[prost(bytes = "vec", tag = "3")] + leader: Vec, + #[prost(uint64, tag = "4")] + epoch: u64, + #[prost(uint32, tag = "5")] + operation: u32, + #[prost(bytes = "vec", repeated, tag = "6")] + frames: Vec>, + #[prost(uint64, tag = "7")] + covered_through: u64, + #[prost(uint64, tag = "8")] + first_sequence: u64, + #[prost(int64, tag = "9")] + deadline_ms: i64, +} + +#[derive(Clone, PartialEq, Message)] +struct ResponseWire { + #[prost(bytes = "vec", tag = "1")] + member: Vec, + #[prost(bytes = "vec", tag = "2")] + request_digest: Vec, + #[prost(uint32, tag = "3")] + status: u32, + #[prost(uint64, tag = "4")] + base_sequence: u64, + #[prost(uint64, tag = "5")] + durable_through: u64, + #[prost(bytes = "vec", repeated, tag = "6")] + frames: Vec>, + #[prost(uint64, optional, tag = "7")] + next_sequence: Option, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum Operation { + Append = 1, + Seal = 2, + Retire = 3, + TailPage = 4, +} + +impl TryFrom for Operation { + type Error = Error; + + fn try_from(value: u32) -> Result { + match value { + 1 => Ok(Self::Append), + 2 => Ok(Self::Seal), + 3 => Ok(Self::Retire), + 4 => Ok(Self::TailPage), + _ => Err(Error::PeerAuthorization("unsupported follower operation")), + } + } +} + +fn session(bytes: &[u8]) -> Result { + let bytes: [u8; 16] = bytes + .try_into() + .map_err(|_| Error::Peer("invalid follower session"))?; + Ok(SessionId::from_bytes(bytes)) +} + +fn node(bytes: &[u8]) -> Result { + let bytes: [u8; 16] = bytes + .try_into() + .map_err(|_| Error::Peer("invalid follower node"))?; + Ok(NodeId::from_bytes(bytes)) +} + +fn validate_request(request: &RequestWire, member: NodeId, now: i64) -> Result { + let operation = Operation::try_from(request.operation)?; + if node(&request.member)? != member + || request.epoch == 0 + || request.deadline_ms <= now + || request.deadline_ms > now.saturating_add(MAX_DEADLINE_AHEAD_MS) + { + return Err(Error::PeerAuthorization( + "follower request scope or deadline is invalid", + )); + } + if operation == Operation::Append + && (request.frames.is_empty() || request.frames.len() > MAX_APPEND_FRAMES) + { + return Err(Error::PeerAuthorization("invalid follower append batch")); + } + if operation != Operation::Append && !request.frames.is_empty() { + return Err(Error::PeerAuthorization("unexpected follower frames")); + } + Ok(operation) +} + +fn signed(body: Vec, key: &SigningKey, domain: &[u8]) -> Vec { + let mut message = Vec::with_capacity(domain.len() + body.len()); + message.extend_from_slice(domain); + message.extend_from_slice(&body); + SignedWire { + body, + signature: key.sign(&message).to_bytes().to_vec(), + } + .encode_to_vec() +} + +fn verify(wire: &[u8], key: ed25519_dalek::VerifyingKey, domain: &[u8]) -> Result> { + let signed = SignedWire::decode(wire).map_err(|_| Error::Peer("invalid follower envelope"))?; + let signature = Signature::from_slice(&signed.signature) + .map_err(|_| Error::PeerAuthorization("invalid follower signature"))?; + let mut message = Vec::with_capacity(domain.len() + signed.body.len()); + message.extend_from_slice(domain); + message.extend_from_slice(&signed.body); + key.verify(&message, &signature) + .map_err(|_| Error::PeerAuthorization("follower signature does not match enrollment"))?; + Ok(signed.body) +} + +async fn follower_address(node: &NodeAdvertisement) -> Result { + let endpoint = node + .endpoint() + .strip_prefix("https://") + .and_then(|address| address.strip_suffix(":8080")) + .ok_or(Error::Peer("follower endpoint is invalid"))?; + tokio::net::lookup_host(format!("{endpoint}:8081")) + .await + .map_err(transport_io)? + .next() + .ok_or(Error::Peer("follower endpoint has no socket address")) +} + +fn transport_io(source: std::io::Error) -> Error { + Error::PeerTransportUnknown { + context: "follower fixture round trip", + source: Box::new(source), + } +} + +async fn receive(socket: &mut TcpStream, maximum: usize) -> Result> { + let mut length = [0; 4]; + socket.read_exact(&mut length).await.map_err(transport_io)?; + let length = u32::from_be_bytes(length) as usize; + if length == 0 || length > maximum { + return Err(Error::Peer("follower message exceeds byte limit")); + } + let mut bytes = vec![0; length]; + socket.read_exact(&mut bytes).await.map_err(transport_io)?; + Ok(bytes) +} + +async fn send(socket: &mut TcpStream, bytes: &[u8], maximum: usize) -> Result<()> { + if bytes.is_empty() || bytes.len() > maximum { + return Err(Error::Peer("follower message exceeds byte limit")); + } + let length = + u32::try_from(bytes.len()).map_err(|_| Error::Peer("follower message too large"))?; + socket + .write_all(&length.to_be_bytes()) + .await + .map_err(transport_io)?; + socket.write_all(bytes).await.map_err(transport_io)?; + Ok(()) +} + +/// One test-only network transport. Every reply is signed by its enrolled member. +pub(super) struct ProcessFollowerTransport { + session: SessionId, + key: SigningKey, + directory: NodeDirectory, + members: StdMutex>, +} + +struct CachedMember { + pinned_session: SessionId, + pinned_key: [u8; 32], + pinned_endpoint: String, + advertisement: NodeAdvertisement, + observed_at: Instant, +} + +impl ProcessFollowerTransport { + pub(super) fn new( + session: SessionId, + key: SigningKey, + directory: NodeDirectory, + members: HashMap, + ) -> Result { + Ok(Self { + session, + key, + directory, + members: StdMutex::new( + members + .into_iter() + .map(|(member, advertisement)| { + Ok(( + member, + CachedMember { + pinned_session: advertisement.session(), + pinned_key: advertisement.verifying_key()?.to_bytes(), + pinned_endpoint: advertisement.endpoint().to_owned(), + advertisement, + observed_at: Instant::now(), + }, + )) + }) + .collect::>>()?, + ), + }) + } + + async fn member(&self, member: NodeId) -> Result { + let (current, pinned_session, pinned_key, pinned_endpoint, observed_at) = { + let members = self + .members + .lock() + .map_err(|_| Error::Peer("follower member cache lock poisoned"))?; + let cached = members + .get(&member) + .ok_or(Error::PeerAuthorization("follower member was not enrolled"))?; + ( + cached.advertisement.clone(), + cached.pinned_session, + cached.pinned_key, + cached.pinned_endpoint.clone(), + cached.observed_at, + ) + }; + let now = now_ms(); + if observed_at.elapsed() < Duration::from_secs(1) + && current.expires_at_ms() > now.saturating_add(1_000) + { + return Ok(current); + } + let fresh = self + .directory + .resolve_node(member, now) + .await? + .ok_or(Error::Fenced)?; + if fresh.session() != pinned_session + || fresh.verifying_key()?.to_bytes() != pinned_key + || fresh.endpoint() != pinned_endpoint + { + return Err(Error::Fenced); + } + let mut members = self + .members + .lock() + .map_err(|_| Error::Peer("follower member cache lock poisoned"))?; + let cached = members + .get_mut(&member) + .ok_or(Error::PeerAuthorization("follower member was not enrolled"))?; + cached.advertisement = fresh.clone(); + cached.observed_at = Instant::now(); + Ok(fresh) + } + + async fn round_trip(&self, member: NodeId, mut request: RequestWire) -> Result { + let enrolled = self.member(member).await?; + if enrolled.node() != member || enrolled.expires_at_ms() <= now_ms() { + return Err(Error::Fenced); + } + request.sender = self.session.as_bytes().to_vec(); + request.member = member.as_bytes().to_vec(); + request.deadline_ms = now_ms() + .checked_add(MAX_DEADLINE_AHEAD_MS) + .ok_or(Error::Deadline)?; + let body = request.encode_to_vec(); + let request_digest = blake3::hash(&body); + let encoded = signed(body, &self.key, REQUEST_DOMAIN); + if encoded.len() > MAX_REQUEST_BYTES { + return Err(Error::Peer("follower request exceeds byte limit")); + } + let address = follower_address(&enrolled).await?; + let response = tokio::time::timeout(REQUEST_TIMEOUT, async { + let mut socket = TcpStream::connect(address).await.map_err(transport_io)?; + send(&mut socket, &encoded, MAX_REQUEST_BYTES).await?; + receive(&mut socket, MAX_RESPONSE_BYTES).await + }) + .await + .map_err(|source| Error::PeerTransportUnknown { + context: "follower fixture deadline", + source: Box::new(source), + })??; + let body = verify(&response, enrolled.verifying_key()?, RESPONSE_DOMAIN)?; + let response = ResponseWire::decode(body.as_slice()) + .map_err(|_| Error::Peer("invalid follower response"))?; + if response.member != member.as_bytes() + || response.request_digest != request_digest.as_bytes() + { + return Err(Error::PeerAuthorization( + "follower response was not bound to request", + )); + } + match response.status { + 0 => Ok(response), + 1 => Err(Error::PeerAuthorization("follower request was refused")), + _ => Err(Error::Peer("follower operation failed")), + } + } +} + +fn request(operation: Operation, leader: SessionId, epoch: u64) -> RequestWire { + RequestWire { + sender: Vec::new(), + member: Vec::new(), + leader: leader.as_bytes().to_vec(), + epoch, + operation: operation as u32, + frames: Vec::new(), + covered_through: 0, + first_sequence: 0, + deadline_ms: 0, + } +} + +impl NodeLogTransport for ProcessFollowerTransport { + fn append<'a>( + &'a self, + member: NodeId, + append: AppendRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + let mut message = request(Operation::Append, append.leader_session, append.log_epoch); + message.frames = append + .frames + .into_iter() + .map(|frame| frame.to_vec()) + .collect(); + message.covered_through = append.covered_through; + let reply = self.round_trip(member, message).await?; + Ok(FollowerReceipt { + base_sequence: reply.base_sequence, + durable_through: reply.durable_through, + }) + }) + } + + fn seal<'a>( + &'a self, + member: NodeId, + seal: SealRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + let reply = self + .round_trip( + member, + request(Operation::Seal, seal.leader_session, seal.log_epoch), + ) + .await?; + Ok(FollowerReceipt { + base_sequence: reply.base_sequence, + durable_through: reply.durable_through, + }) + }) + } + + fn retire<'a>( + &'a self, + member: NodeId, + retire: RetireRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + let mut message = request(Operation::Retire, retire.leader_session, retire.log_epoch); + message.covered_through = retire.covered_through; + let reply = self.round_trip(member, message).await?; + Ok(FollowerReceipt { + base_sequence: reply.base_sequence, + durable_through: reply.durable_through, + }) + }) + } + + fn tail<'a>(&'a self, member: NodeId, tail: TailRequest) -> BoxFuture<'a, Result>> { + Box::pin(async move { + let page = self.tail_page(member, tail).await?; + if page.next_sequence.is_some() { + return Err(Error::Peer("unbounded follower tail is unsupported")); + } + Ok(page.frames) + }) + } + + fn tail_page<'a>( + &'a self, + member: NodeId, + tail: TailRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + let mut message = request(Operation::TailPage, tail.leader_session, tail.log_epoch); + message.first_sequence = tail.first_sequence; + let reply = self.round_trip(member, message).await?; + Ok(FollowerTailPage { + frames: reply.frames.into_iter().map(Bytes::from).collect(), + next_sequence: reply.next_sequence, + }) + }) + } +} + +/// Serves one node's durable store on the private Compose network. +pub(super) fn serve( + listener: TcpListener, + member: NodeId, + store: FollowerStore, + directory: NodeDirectory, + key: SigningKey, +) -> tokio::task::JoinHandle<()> { + tokio::spawn(async move { + while let Ok((socket, _)) = listener.accept().await { + let store = store.clone(); + let directory = directory.clone(); + let key = key.clone(); + tokio::spawn(async move { + let _ = tokio::time::timeout( + REQUEST_TIMEOUT, + serve_one(socket, member, store, directory, key), + ) + .await; + }); + } + }) +} + +async fn serve_one( + mut socket: TcpStream, + member: NodeId, + store: FollowerStore, + directory: NodeDirectory, + key: SigningKey, +) -> Result<()> { + let encoded = receive(&mut socket, MAX_REQUEST_BYTES).await?; + let envelope = SignedWire::decode(encoded.as_slice()) + .map_err(|_| Error::Peer("invalid follower envelope"))?; + let request = RequestWire::decode(envelope.body.as_slice()) + .map_err(|_| Error::Peer("invalid follower request"))?; + let sender = session(&request.sender)?; + let leader = session(&request.leader)?; + validate_request(&request, member, now_ms())?; + let enrolled = directory + .load(sender, now_ms()) + .await? + .ok_or(Error::PeerAuthorization("follower sender is not live"))?; + verify( + &encoded, + enrolled.advertisement().verifying_key()?, + REQUEST_DOMAIN, + )?; + let digest = blake3::hash(&envelope.body); + let result = execute(member, store, directory, sender, leader, request).await; + let mut reply = ResponseWire { + member: member.as_bytes().to_vec(), + request_digest: digest.as_bytes().to_vec(), + status: 0, + base_sequence: 0, + durable_through: 0, + frames: Vec::new(), + next_sequence: None, + }; + match result { + Ok(Reply::Receipt(receipt)) => { + reply.base_sequence = receipt.base_sequence; + reply.durable_through = receipt.durable_through; + } + Ok(Reply::Page(page)) => { + reply.frames = page + .frames + .into_iter() + .map(|frame| frame.to_vec()) + .collect(); + reply.next_sequence = page.next_sequence; + } + Err(Error::PeerAuthorization(_)) | Err(Error::Fenced) => reply.status = 1, + Err(_) => reply.status = 2, + } + let encoded = signed(reply.encode_to_vec(), &key, RESPONSE_DOMAIN); + send(&mut socket, &encoded, MAX_RESPONSE_BYTES).await +} + +enum Reply { + Receipt(FollowerReceipt), + Page(FollowerTailPage), +} + +async fn execute( + member: NodeId, + store: FollowerStore, + directory: NodeDirectory, + sender: SessionId, + leader: SessionId, + request: RequestWire, +) -> Result { + match Operation::try_from(request.operation)? { + Operation::Append => { + if sender != leader { + return Err(Error::PeerAuthorization( + "invalid follower append sender or batch", + )); + } + directory + .authorize_log_append( + leader, + member, + request.epoch, + request.covered_through, + now_ms(), + ) + .await?; + store + .append( + leader, + request.epoch, + request.frames.into_iter().map(Bytes::from).collect(), + request.covered_through, + ) + .await + .map(Reply::Receipt) + } + Operation::Retire => { + if sender != leader { + return Err(Error::PeerAuthorization( + "invalid follower retirement sender", + )); + } + directory + .authorize_log_retire( + leader, + member, + request.epoch, + request.covered_through, + now_ms(), + ) + .await?; + store + .retire(leader, request.epoch, request.covered_through) + .await + .map(Reply::Receipt) + } + Operation::Seal | Operation::TailPage => { + directory + .authorize_log_recovery(leader, sender, member, request.epoch, now_ms()) + .await?; + if request.operation == Operation::Seal as u32 { + store.seal(leader, request.epoch).await.map(Reply::Receipt) + } else { + store + .read_tail_page(leader, request.epoch, request.first_sequence) + .await + .map(Reply::Page) + } + } + } +} + +/// One serialized authority view shared by heartbeats, enrollment, and log CAS. +#[derive(Clone)] +pub(super) struct ProcessEnrollment { + pub(super) observed: Arc>, + directory: NodeDirectory, + session: SessionId, +} + +impl ProcessEnrollment { + pub(super) fn new( + observed: VersionedNodeAdvertisement, + directory: NodeDirectory, + session: SessionId, + ) -> Self { + Self { + observed: Arc::new(Mutex::new(observed)), + directory, + session, + } + } + + pub(super) async fn refresh(&self, next: NodeAdvertisement) -> Result { + let mut observed = self.observed.lock().await; + *observed = self.directory.refresh(&observed, next, now_ms()).await?; + Ok(observed.advertisement().expires_at_ms()) + } + + pub(super) async fn withdraw(&self) -> Result<()> { + let observed = self.observed.lock().await; + self.directory + .withdraw_after_drain(&observed, now_ms()) + .await + } + + pub(super) async fn progress(&self) -> u64 { + self.observed.lock().await.advertisement().progress() + } + + pub(super) async fn generation(&self) -> u64 { + self.observed.lock().await.advertisement().generation() + } +} + +impl NodeLogAuthority for ProcessEnrollment { + fn activate<'a>(&'a self, epoch: u64) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + let mut observed = self.observed.lock().await; + if observed + .advertisement() + .log() + .is_none_or(|log| log.epoch() != epoch) + { + return Err(Error::Fenced); + } + *observed = self.directory.activate_log(&observed, now_ms()).await?; + Ok(()) + }) + } + + fn advance_coverage<'a>(&'a self, epoch: u64, through: u64) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + let mut observed = self.observed.lock().await; + if observed + .advertisement() + .log() + .is_none_or(|log| log.epoch() != epoch) + { + return Err(Error::Fenced); + } + *observed = self + .directory + .advance_log_coverage(&observed, through, now_ms()) + .await?; + Ok(()) + }) + } + + fn close<'a>(&'a self, barrier: &'a NodeLogRotationBarrier) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { + let mut observed = self.observed.lock().await; + if observed + .advertisement() + .log() + .is_none_or(|log| log.epoch() != barrier.log_epoch()) + { + return Err(Error::Fenced); + } + *observed = self + .directory + .close_log(&observed, barrier, now_ms()) + .await?; + Ok(()) + }) + } +} + +pub(super) struct ProcessDurabilityProvider { + enrollment: ProcessEnrollment, + key: SigningKey, + lease: NodeLeaseGuard, + telemetry: CellTelemetryHandle, +} + +impl ProcessDurabilityProvider { + pub(super) fn new( + enrollment: ProcessEnrollment, + key: SigningKey, + lease: NodeLeaseGuard, + telemetry: CellTelemetryHandle, + ) -> Self { + Self { + enrollment, + key, + lease, + telemetry, + } + } +} + +impl NodeDurabilityProvider for ProcessDurabilityProvider { + fn recruit( + self: Arc, + limits: Limits, + required_follower_bytes: u64, + live_node_limit: usize, + ) -> Pin>> + Send>> { + Box::pin(async move { + let result: Result> = async { + let mut observed = self.enrollment.observed.lock().await; + if observed.advertisement().log().is_none() { + // Keep the measured ensemble shape fixed at both followers. + if self + .enrollment + .directory + .live(now_ms(), live_node_limit) + .await? + .len() + < 3 + { + return Ok(None); + } + let next = self + .enrollment + .directory + .try_recruit_log( + &observed, + 1, + required_follower_bytes, + live_node_limit, + now_ms(), + ) + .await?; + let Some(next) = next else { + return Ok(None); + }; + *observed = next; + } + let log = observed + .advertisement() + .log() + .ok_or(Error::Node("enrolled follower log disappeared"))?; + let members = log.members().to_vec(); + if members.len() != 2 { + return Err(Error::Node("capacity lane requires two follower members")); + } + let epoch = log.epoch(); + let mut enrolled = HashMap::new(); + for &member in &members { + let advertisement = self + .enrollment + .directory + .resolve_node(member, now_ms()) + .await? + .ok_or(Error::Node("follower member is no longer live"))?; + enrolled.insert(member, advertisement); + } + let transport: Arc = Arc::new(ProcessFollowerTransport::new( + self.enrollment.session, + self.key.clone(), + self.enrollment.directory.clone(), + enrolled, + )?); + let authority: Arc = Arc::new(self.enrollment.clone()); + NodeDurabilityConfig::new( + self.enrollment.session, + NodeId::from_bytes(*self.enrollment.session.as_bytes()), + epoch, + members, + transport, + authority, + self.lease.clone(), + limits, + self.telemetry.clone(), + ) + .map(Some) + } + .await; + result.map_err(|source| Box::new(source) as Box) + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn signed_envelope_binds_sender_key_domain_and_body() { + let key = SigningKey::from_bytes(&[7; 32]); + let other = SigningKey::from_bytes(&[8; 32]); + let encoded = signed(vec![1, 2, 3], &key, REQUEST_DOMAIN); + assert_eq!( + verify(&encoded, key.verifying_key(), REQUEST_DOMAIN).unwrap(), + vec![1, 2, 3] + ); + assert!(verify(&encoded, other.verifying_key(), REQUEST_DOMAIN).is_err()); + assert!(verify(&encoded, key.verifying_key(), RESPONSE_DOMAIN).is_err()); + let mut changed = SignedWire::decode(encoded.as_slice()).unwrap(); + changed.body.push(4); + assert!( + verify( + &changed.encode_to_vec(), + key.verifying_key(), + REQUEST_DOMAIN + ) + .is_err() + ); + } + + #[test] + fn request_scope_rejects_wrong_member_stale_epoch_deadline_and_batch() { + let member = NodeId::from_bytes([1; 16]); + let leader = SessionId::from_bytes([2; 16]); + let mut append = request(Operation::Append, leader, 1); + append.member = member.as_bytes().to_vec(); + append.frames.push(vec![1]); + append.deadline_ms = 11_000; + assert_eq!( + validate_request(&append, member, 10_000).unwrap(), + Operation::Append + ); + append.member = NodeId::from_bytes([3; 16]).as_bytes().to_vec(); + assert!(validate_request(&append, member, 10_000).is_err()); + append.member = member.as_bytes().to_vec(); + append.epoch = 0; + assert!(validate_request(&append, member, 10_000).is_err()); + append.epoch = 1; + append.deadline_ms = 10_000; + assert!(validate_request(&append, member, 10_000).is_err()); + append.deadline_ms = 20_001; + assert!(validate_request(&append, member, 10_000).is_err()); + append.deadline_ms = 11_000; + append.frames = vec![vec![1]; MAX_APPEND_FRAMES + 1]; + assert!(validate_request(&append, member, 10_000).is_err()); + append.operation = Operation::Retire as u32; + assert!(validate_request(&append, member, 10_000).is_err()); + } + + #[tokio::test] + async fn oversized_message_is_rejected_before_network_write() { + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let mut socket = TcpStream::connect(listener.local_addr().unwrap()) + .await + .unwrap(); + assert!( + send( + &mut socket, + &vec![1; MAX_REQUEST_BYTES + 1], + MAX_REQUEST_BYTES + ) + .await + .is_err() + ); + } +} diff --git a/crates/cellule-app/tests/process_node.rs b/crates/cellule-app/tests/process_node.rs index 0ef3b9c..9d0c698 100644 --- a/crates/cellule-app/tests/process_node.rs +++ b/crates/cellule-app/tests/process_node.rs @@ -1,8 +1,9 @@ //! Public host and authoritative session lifecycle for the process fixture. use super::performance_fixture::{DurabilityRecorder, node_session, now_ms}; +use super::process_follower::{ProcessDurabilityProvider, ProcessEnrollment}; use crate::*; -use cellule_host::{CellNode, CellNodeBuilder}; +use cellule_host::{CellNode, CellNodeBuilder, NodeDurabilitySupervisorConfig}; use cellule_runtime::node::lease::NodeLeaseGuard; use std::time::Duration; use tokio_util::sync::CancellationToken; @@ -16,6 +17,12 @@ pub(super) fn directory(layout: &CellStorageLayout, registry: &Registry) -> Node ) } +pub(super) fn signing_key(node: usize) -> SigningKey { + let mut seed = [93; 32]; + seed[0] = u8::try_from(node).unwrap(); + SigningKey::from_bytes(&seed) +} + pub(super) async fn start( node: usize, application: Arc, @@ -26,6 +33,21 @@ pub(super) async fn start( CellNode, Arc, cellule_host::read_replicas::ReadReplicaManager, +) { + start_configured(node, application, layout, root, endpoint, false).await +} + +pub(super) async fn start_configured( + node: usize, + application: Arc, + layout: &CellStorageLayout, + root: &std::path::Path, + endpoint: String, + follower_enabled: bool, +) -> ( + CellNode, + Arc, + cellule_host::read_replicas::ReadReplicaManager, ) { let registry = application.registry(); // Keep writer admission at 32 Cells while charging both the old and new @@ -34,16 +56,23 @@ pub(super) async fn start( .unwrap() .with_native_memory_limit(32 << 20) .unwrap(); - let host = CellNodeBuilder::new(application) + let mut builder = CellNodeBuilder::new(application) .with_runtime(pool, 64 * 1024 * 1024) .with_session(node_session(node)) - .with_replica_host(reference_host()) - .build() - .unwrap(); + .with_replica_host(reference_host()); + if follower_enabled { + builder = builder.with_follower_store( + root.join("followers"), + Limits::default(), + DiskBudget::new(1 << 30), + ); + } + let host = builder.build().unwrap(); let durability = Arc::new(DurabilityRecorder::default()); host.install_telemetry(durability.clone()).unwrap(); let directory = directory(layout, ®istry); - let signer = SigningKey::from_bytes(&[93; 32]); + let signer = signing_key(node); + let follower_signer = signer.clone(); let advertisement = move |now: i64, progress| { NodeAdvertisement::sign( NodeId::from_bytes(*node_session(node).as_bytes()), @@ -64,22 +93,50 @@ pub(super) async fn start( free_memory_bytes: 64 * 1024 * 1024, free_disk_bytes: 1 << 30, job_credits: 32, + follower_free_bytes: if follower_enabled { 1 << 30 } else { 0 }, + log_protocol: if follower_enabled { + cellule_runtime::node::NODE_LOG_PROTOCOL_VERSION + } else { + 0 + }, ..NodeCapacity::default() }, ) }; let now = now_ms(); - let mut observed = directory + let observed = directory .create(advertisement(now, 1).unwrap(), now) .await .unwrap(); let lease = NodeLeaseGuard::new(now_ms(), observed.advertisement().expires_at_ms()).unwrap(); + let enrollment = ProcessEnrollment::new(observed, directory.clone(), node_session(node)); let shutdown = CancellationToken::new(); let tasks = host .install_task_group(CancellationToken::new(), shutdown.clone()) .unwrap(); host.install_node_lease_for_startup(lease.clone()).unwrap(); + if follower_enabled { + let provider = Arc::new(ProcessDurabilityProvider::new( + enrollment.clone(), + follower_signer, + lease.clone(), + host.runtime().telemetry_handle(), + )); + let configuration = NodeDurabilitySupervisorConfig::new( + ApplicationId::from_bytes([82; 16]), + Limits::default(), + 1 << 20, + 32, + Duration::from_millis(250), + Duration::from_secs(1), + 1_000_000, + ) + .unwrap(); + host.install_node_durability_provider(provider, configuration) + .unwrap(); + } let (renewed, first_renewal) = tokio::sync::oneshot::channel(); + let renewing = enrollment.clone(); tasks .spawn_lease_maintenance(async move { let mut renewed = Some(renewed); @@ -90,29 +147,24 @@ pub(super) async fn start( () = tokio::time::sleep(Duration::from_secs(5)) => {} } let now = now_ms(); - let next = advertisement(now, observed.advertisement().progress() + 1)?; + let next = advertisement(now, renewing.progress().await + 1)?; // Finish the conditional write before observing shutdown so // withdrawal always uses the latest acknowledged generation. - observed = tokio::time::timeout( - lease.remaining(), - directory.refresh(&observed, next, now), - ) - .await - .map_err(|_| Error::Deadline)??; - lease.renew(now_ms(), observed.advertisement().expires_at_ms())?; + let expires_at = + tokio::time::timeout(lease.remaining(), renewing.refresh(next)) + .await + .map_err(|_| Error::Deadline)??; + lease.renew(now_ms(), expires_at)?; if let Some(renewed) = renewed.take() { let _ = renewed.send(()); } } - tokio::time::timeout( - Duration::from_secs(20), - directory.withdraw(&observed, now_ms()), - ) - .await - .map_err(|_| Error::Deadline)??; + tokio::time::timeout(Duration::from_secs(20), renewing.withdraw()) + .await + .map_err(|_| Error::Deadline)??; println!( "PERF node_{node}_session_withdrawn: generation={}", - observed.advertisement().generation() + renewing.generation().await ); Ok(()) } From c63341cafa0ed937652c6cc3198664b1702f8380 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 20:16:15 -0700 Subject: [PATCH 10/14] Validate follower recovery and measure network append time --- .../performance/2026-09-29-write-capacity.md | 54 +++ crates/cellule-app/qualification/entities.py | 18 +- .../qualification/test_entities.py | 12 +- crates/cellule-app/tests/entities/process.rs | 5 + .../tests/entities/process/observation.rs | 16 + .../cellule-app/tests/performance_fixture.rs | 17 + crates/cellule-app/tests/process_follower.rs | 308 +++++++++++++++++- crates/cellule-app/tests/process_node.rs | 3 +- 8 files changed, 424 insertions(+), 9 deletions(-) diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index 57c0357..e74ddbb 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -225,6 +225,60 @@ and `bfbef7176b533f326d1e536122a814afbf6e2b7d8bbcfe3184574fcbac0b4940`. The downloaded artifact is under `$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36659182959`. +## First networked follower-proof comparison + +[CI run 36662251677](https://github.com/crabbuild/cellule/actions/runs/36662251677) +passed both capacity jobs, each with three fresh-provider repeats and the same +release binary (`69ad461750472d181e1a326a2d4f5e6ef8ade22ca14920e9b4c4e40ad83bf45f`). +The test-merge source was `f2a5db377cbc11da01278914bc6fa1996b4b4565`. +The follower lane used a signed TCP endpoint, two private-disk followers per +owner, and the object-only lane's 12 Cells, arrivals, resource limits, and +readback rules. Every acknowledged write passed readback, and final published +roots covered all receipts. Follower response winners were Fleet/Object +2,245/24, 3,740/7, and 4,657/24 across repeats 1–3. The object lane used +only Object proofs. + +| Shape | Object fully served offered rate, repeats 1–3 | Follower fully served offered rate, repeats 1–3 | Follower first overloaded rate | +| --- | --- | --- | --- | +| Uniform writes | 16, 32, 32 actions/node/s | 4, 24, 24 actions/node/s | 16, 32, 32 | +| One hot writable Cell | 24, 16, 24 | 16, 4, 16 | 24, 16, 24 | +| Skewed, 20% writes | 48, 64, 64 | 24, 32, 48 | 32, 48, 64 | + +At the common hot rate of 16 actions/node/s, all object repeats were fully +served with action p95 of 48.06–56.90 ms. Follower repeats 1 and 3 were fully +served with p95 of 17.12–26.37 ms; repeat 2 missed four scheduled arrivals +despite p95 of 21.58 ms among successful actions. Hot follower overload at +24 actions/node/s returned 168–238 owner capacity refusals in repeats 1 and +3. The follower path reduces response latency at this matched point, but its +fully served throughput bound is lower and variable on these shared runners. +The two jobs used different runners, so repeat numbers are not paired host +measurements. + +The response and publication capacities differ. A fully served hot follower +window in repeat 1 completed about 48 writes/s while publishing 46.56 roots/s +and ended 13 commits ahead of its root. Its final root did drain and cover +every acknowledged receipt. Uniform fully served windows in repeats 2 and 3 +ended one commit ahead. These are response-supported windows; their published +roots/s and remaining lag must be read separately. The first overloaded +uniform and skewed follower points were mostly isolated scheduler-late +arrivals, so they do not establish storage saturation. + +The three follower `verification.json` SHA-256 digests are +`c927b09f8f0200b05548d02827aba28434014b8fa1dc7cea0e983f0893001e6e`, +`2e2e9a657fc200103224f2830804f9f9d629b390b91500498800cb74f38ee8b7`, +and `12f93bc42bd015f081f528504ff82f580edfe8db3c78eef7b5d1a52e53d6f75c`. +The corresponding object digests are +`7ddbbf67f22b8554738c73b73d2be76338f52deb46bb9223f1cda83dca22edca`, +`2cf37c2a34158f9e79a58850fc64873924642149fdd83b7e2d9d326adc90a24f`, +and `64c9182aa10c691a8ae32da608b9533f8a0384b02f08f6c9e8e0d3e438f83ac9`. +Raw logs, samples, provider digests, and these reports are under +`$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36662251677`. +The enclosing PR check failed in its separate reader smoke because that +fixture still used a shared signing key after advertisements switched to +per-node keys. The next revision corrects that mismatch and adds network +append duration evidence; this capacity result remains valid for its pinned +binary and must be followed by a clean full rerun. + After preparing a fresh source/binary snapshot with the Compose qualification guide, run each repeat with a new state directory and Compose project: diff --git a/crates/cellule-app/qualification/entities.py b/crates/cellule-app/qualification/entities.py index 40f49e5..835c46b 100644 --- a/crates/cellule-app/qualification/entities.py +++ b/crates/cellule-app/qualification/entities.py @@ -62,6 +62,7 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic captures = rows(control / f"node-{node}-captures.tsv") costs = rows(control / f"node-{node}-publication-costs.tsv") appends = rows(control / f"node-{node}-follower-appends.tsv") + network = rows(control / f"node-{node}-follower-network.tsv") assert responses, f"node {node}: missing command response evidence" assert executions, f"node {node}: missing command execution evidence" assert publications, f"node {node}: missing publication evidence" @@ -102,6 +103,10 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic for row in appends: assert int(row["at_ms"]) > 0 and int(row["bytes"]) >= 0 assert row["acknowledged"] in {"true", "false"} + for row in network: + assert int(row["at_ms"]) > 0 + assert int(row["bytes"]) >= 0 and int(row["duration_us"]) >= 0 + assert row["acknowledged"] in {"true", "false"} for window in windows: if node >= window["nodes"]: continue @@ -113,6 +118,7 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic selected_captures = [row for row in captures if start <= int(row["at_ms"]) <= end] selected_costs = [row for row in costs if start <= int(row["at_ms"]) <= end] selected_appends = [row for row in appends if start <= int(row["at_ms"]) <= end] + selected_network = [row for row in network if start <= int(row["at_ms"]) <= end] response_sources = {source: sum(row["source"] == source for row in selected_responses) for source in sorted(sources)} window.setdefault("node_durability", {})[node] = dict( @@ -143,7 +149,9 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic uploaded_bytes=sum(int(row["bytes"]) for row in selected_costs), follower_appends=len(selected_appends), follower_append_failures=sum(row["acknowledged"] == "false" for row in selected_appends), - follower_append_bytes=sum(int(row["bytes"]) for row in selected_appends)) + follower_append_bytes=sum(int(row["bytes"]) for row in selected_appends), + follower_network_latency=distribution([int(row["duration_us"]) for row in selected_network]), + follower_network_bytes=sum(int(row["bytes"]) for row in selected_network)) return dict(response_sources={source: sum(row["source"] == source for row in responses) for source in sorted(sources)}, response_latency=distribution([int(row["response_us"]) for row in responses]), @@ -156,6 +164,8 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic follower_appends=len(appends), acknowledged_follower_appends=sum(row["acknowledged"] == "true" for row in appends), follower_append_bytes=sum(int(row["bytes"]) for row in appends), + acknowledged_network_appends=sum(row["acknowledged"] == "true" for row in network), + follower_network_latency=distribution([int(row["duration_us"]) for row in network]), completed_publications=sum(row["succeeded"] == "true" for row in publications), failed_publications=sum(row["succeeded"] == "false" for row in publications)) @@ -279,9 +289,13 @@ def verify_follower_proof(resources: dict) -> dict: for resource in resources.values()) acknowledged_appends = sum(resource["durability"]["acknowledged_follower_appends"] for resource in resources.values()) + network_appends = sum(resource["durability"]["acknowledged_network_appends"] + for resource in resources.values()) assert fleet_proofs > 0, "follower lane returned no follower-proof responses" assert acknowledged_appends > 0, "missing acknowledged follower append evidence" - return dict(follower_proof_responses=fleet_proofs, follower_appends=acknowledged_appends) + assert network_appends > 0, "missing network follower append evidence" + return dict(follower_proof_responses=fleet_proofs, follower_appends=acknowledged_appends, + network_follower_appends=network_appends) def verify_entities(control: Path, capacity: bool = False, follower: bool = False) -> dict: diff --git a/crates/cellule-app/qualification/test_entities.py b/crates/cellule-app/qualification/test_entities.py index 6dfa3f2..0773faf 100644 --- a/crates/cellule-app/qualification/test_entities.py +++ b/crates/cellule-app/qualification/test_entities.py @@ -99,6 +99,8 @@ def setUp(self): "at_ms\tobjects\tbytes\n100003\t2\t2048\n") (self.root / "node-0-follower-appends.tsv").write_text( "at_ms\tacknowledged\tbytes\n") + (self.root / "node-0-follower-network.tsv").write_text( + "at_ms\tacknowledged\tbytes\tduration_us\n") self.windows = [dict(nodes=3, started_ms=100000, ended_ms=110000, elapsed_us=10_000_000)] def test_response_winner_and_later_publication_are_separate(self): @@ -223,7 +225,8 @@ def test_missing_provider_operation_is_rejected(self): class FollowerProofEvidence(unittest.TestCase): def test_follower_lane_requires_proof_and_acknowledged_append(self): resources = {0: dict(durability=dict(response_sources=dict(Fleet=0), - acknowledged_follower_appends=1))} + acknowledged_follower_appends=1, + acknowledged_network_appends=1))} with self.assertRaisesRegex(AssertionError, "no follower-proof responses"): verify_follower_proof(resources) resources[0]["durability"]["response_sources"]["Fleet"] = 1 @@ -231,8 +234,13 @@ def test_follower_lane_requires_proof_and_acknowledged_append(self): with self.assertRaisesRegex(AssertionError, "missing acknowledged follower append"): verify_follower_proof(resources) resources[0]["durability"]["acknowledged_follower_appends"] = 2 + resources[0]["durability"]["acknowledged_network_appends"] = 0 + with self.assertRaisesRegex(AssertionError, "missing network follower append"): + verify_follower_proof(resources) + resources[0]["durability"]["acknowledged_network_appends"] = 2 self.assertEqual(verify_follower_proof(resources), - dict(follower_proof_responses=1, follower_appends=2)) + dict(follower_proof_responses=1, follower_appends=2, + network_follower_appends=2)) if __name__ == "__main__": diff --git a/crates/cellule-app/tests/entities/process.rs b/crates/cellule-app/tests/entities/process.rs index 6e9ee74..b3d1745 100644 --- a/crates/cellule-app/tests/entities/process.rs +++ b/crates/cellule-app/tests/entities/process.rs @@ -206,6 +206,11 @@ async fn entity_process_node() { server.abort(); let _ = server.await; if let Some(server) = follower_server { + // Other owners can still be retiring their lanes during concurrent + // shutdown; keep every follower listener available until all drain. + for peer in 0..3 { + wait_for_marker(&sync.join(format!("node-{peer}.done"))).await; + } server.abort(); let _ = server.await; } diff --git a/crates/cellule-app/tests/entities/process/observation.rs b/crates/cellule-app/tests/entities/process/observation.rs index b84c69c..43c99e3 100644 --- a/crates/cellule-app/tests/entities/process/observation.rs +++ b/crates/cellule-app/tests/entities/process/observation.rs @@ -51,6 +51,7 @@ pub(super) struct NodeObservations { captures: BufWriter, publication_costs: BufWriter, follower_appends: BufWriter, + follower_network: BufWriter, } impl NodeObservations { @@ -72,6 +73,7 @@ impl NodeObservations { captures: create("captures"), publication_costs: create("publication-costs"), follower_appends: create("follower-appends"), + follower_network: create("follower-network"), } } @@ -250,6 +252,19 @@ impl NodeObservations { for (at_ms, acknowledged, bytes) in durability.follower_appends() { writeln!(self.follower_appends, "{at_ms}\t{acknowledged}\t{bytes}").unwrap(); } + writeln!( + self.follower_network, + "at_ms\tacknowledged\tbytes\tduration_us" + ) + .unwrap(); + for (at_ms, acknowledged, bytes, elapsed) in durability.follower_network() { + writeln!( + self.follower_network, + "{at_ms}\t{acknowledged}\t{bytes}\t{}", + elapsed.as_micros() + ) + .unwrap(); + } self.objects.flush().unwrap(); self.object_operations.flush().unwrap(); self.durability.flush().unwrap(); @@ -260,6 +275,7 @@ impl NodeObservations { self.captures.flush().unwrap(); self.publication_costs.flush().unwrap(); self.follower_appends.flush().unwrap(); + self.follower_network.flush().unwrap(); } } diff --git a/crates/cellule-app/tests/performance_fixture.rs b/crates/cellule-app/tests/performance_fixture.rs index 9241790..e102c3c 100644 --- a/crates/cellule-app/tests/performance_fixture.rs +++ b/crates/cellule-app/tests/performance_fixture.rs @@ -154,6 +154,7 @@ pub(super) struct DurabilityRecorder { captures: Mutex>, publication_costs: Mutex>, follower_appends: Mutex>, + follower_network: Mutex>, } impl DurabilityRecorder { @@ -193,6 +194,22 @@ impl DurabilityRecorder { pub(super) fn follower_appends(&self) -> Vec<(i64, bool, u64)> { self.follower_appends.lock().unwrap().clone() } + + pub(super) fn record_follower_network( + &self, + acknowledged: bool, + bytes: u64, + elapsed: Duration, + ) { + self.follower_network + .lock() + .unwrap() + .push((now_ms(), acknowledged, bytes, elapsed)); + } + + pub(super) fn follower_network(&self) -> Vec<(i64, bool, u64, Duration)> { + self.follower_network.lock().unwrap().clone() + } } impl CellTelemetry for DurabilityRecorder { diff --git a/crates/cellule-app/tests/process_follower.rs b/crates/cellule-app/tests/process_follower.rs index d956e4c..0bc73f5 100644 --- a/crates/cellule-app/tests/process_follower.rs +++ b/crates/cellule-app/tests/process_follower.rs @@ -1,6 +1,6 @@ //! Test-only authenticated network transport for private-disk follower lanes. -use super::performance_fixture::now_ms; +use super::performance_fixture::{DurabilityRecorder, now_ms}; use bytes::Bytes; use cellule_host::{FacilityResult, NodeDurabilityProvider}; use cellule_ltx::Limits; @@ -29,7 +29,7 @@ use std::{ use tokio::{ io::{AsyncReadExt, AsyncWriteExt}, net::{TcpListener, TcpStream}, - sync::Mutex, + sync::{Mutex, Semaphore}, }; const REQUEST_DOMAIN: &[u8] = b"cellule.test-follower.request.v1\0"; @@ -173,9 +173,16 @@ async fn follower_address(node: &NodeAdvertisement) -> Result { let endpoint = node .endpoint() .strip_prefix("https://") - .and_then(|address| address.strip_suffix(":8080")) .ok_or(Error::Peer("follower endpoint is invalid"))?; - tokio::net::lookup_host(format!("{endpoint}:8081")) + let (host, gateway_port) = endpoint + .rsplit_once(':') + .ok_or(Error::Peer("follower endpoint has no gateway port"))?; + let follower_port = gateway_port + .parse::() + .ok() + .and_then(|port| port.checked_add(1)) + .ok_or(Error::Peer("follower endpoint port is invalid"))?; + tokio::net::lookup_host(format!("{host}:{follower_port}")) .await .map_err(transport_io)? .next() @@ -221,6 +228,7 @@ pub(super) struct ProcessFollowerTransport { key: SigningKey, directory: NodeDirectory, members: StdMutex>, + observation: Option>, } struct CachedMember { @@ -237,11 +245,13 @@ impl ProcessFollowerTransport { key: SigningKey, directory: NodeDirectory, members: HashMap, + observation: Option>, ) -> Result { Ok(Self { session, key, directory, + observation, members: StdMutex::new( members .into_iter() @@ -309,6 +319,25 @@ impl ProcessFollowerTransport { } async fn round_trip(&self, member: NodeId, mut request: RequestWire) -> Result { + let append = request.operation == Operation::Append as u32; + let append_bytes = request.frames.iter().map(Vec::len).sum::(); + let started = Instant::now(); + let result = self.round_trip_inner(member, &mut request).await; + if append && let Some(observation) = &self.observation { + observation.record_follower_network( + result.is_ok(), + u64::try_from(append_bytes).unwrap_or(u64::MAX), + started.elapsed(), + ); + } + result + } + + async fn round_trip_inner( + &self, + member: NodeId, + request: &mut RequestWire, + ) -> Result { let enrolled = self.member(member).await?; if enrolled.node() != member || enrolled.expires_at_ms() <= now_ms() { return Err(Error::Fenced); @@ -460,11 +489,16 @@ pub(super) fn serve( key: SigningKey, ) -> tokio::task::JoinHandle<()> { tokio::spawn(async move { + let slots = Arc::new(Semaphore::new(512)); while let Ok((socket, _)) = listener.accept().await { + let Ok(slot) = Arc::clone(&slots).try_acquire_owned() else { + continue; + }; let store = store.clone(); let directory = directory.clone(); let key = key.clone(); tokio::spawn(async move { + let _slot = slot; let _ = tokio::time::timeout( REQUEST_TIMEOUT, serve_one(socket, member, store, directory, key), @@ -706,6 +740,7 @@ pub(super) struct ProcessDurabilityProvider { key: SigningKey, lease: NodeLeaseGuard, telemetry: CellTelemetryHandle, + observation: Arc, } impl ProcessDurabilityProvider { @@ -714,12 +749,14 @@ impl ProcessDurabilityProvider { key: SigningKey, lease: NodeLeaseGuard, telemetry: CellTelemetryHandle, + observation: Arc, ) -> Self { Self { enrollment, key, lease, telemetry, + observation, } } } @@ -786,6 +823,7 @@ impl NodeDurabilityProvider for ProcessDurabilityProvider { self.key.clone(), self.enrollment.directory.clone(), enrolled, + Some(Arc::clone(&self.observation)), )?); let authority: Arc = Arc::new(self.enrollment.clone()); NodeDurabilityConfig::new( @@ -810,6 +848,83 @@ impl NodeDurabilityProvider for ProcessDurabilityProvider { #[cfg(test)] mod tests { use super::*; + use cellule_ltx::{Db, NodeFrameScope, encode_node_frame}; + use cellule_runtime::identity::Digest; + use cellule_runtime::ltx::CellStorageLayout; + use cellule_runtime::node::log_recovery::NodeLogRecovery; + use cellule_runtime::node::{NODE_LOG_PROTOCOL_VERSION, NodeCapacity, NodeFailureDomain}; + use cellule_store::Store; + use object_store::{memory::InMemory, path::Path}; + + fn advertisement( + node: NodeId, + session: SessionId, + key: &SigningKey, + endpoint: String, + issued_at: i64, + lifetime_ms: i64, + ) -> NodeAdvertisement { + NodeAdvertisement::sign( + node, + session, + endpoint, + Digest::from_bytes([10; 32]), + Digest::from_bytes([11; 32]), + Digest::from_bytes([12; 32]), + Digest::from_bytes([13; 32]), + key, + 1, + issued_at, + issued_at + lifetime_ms, + vec![Digest::from_bytes([14; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1 << 20, + free_disk_bytes: 1 << 20, + follower_free_bytes: 1 << 20, + job_credits: 4, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + ..NodeCapacity::default() + }, + ) + .unwrap() + } + + fn frame(limits: Limits) -> Bytes { + let source = tempfile::TempDir::new().unwrap(); + let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); + database + .transaction(|transaction| { + transaction.execute_batch( + "CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ + INSERT INTO events(body) VALUES ('survives')", + ) + }) + .unwrap(); + let capture = database.capture().unwrap(); + let segment = capture.segments.first().unwrap(); + let encoded = encode_node_frame( + NodeFrameScope { + leader_session: [1; 16], + log_epoch: 1, + node_sequence: 1, + application: [3; 16], + cell: [4; 32], + incarnation: [5; 16], + cell_epoch: 1, + commit_sequence: 1, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + limits, + ) + .unwrap() + .encoded() + .clone(); + database.close().unwrap(); + encoded + } #[test] fn signed_envelope_binds_sender_key_domain_and_body() { @@ -879,4 +994,189 @@ mod tests { .is_err() ); } + + #[tokio::test] + async fn owner_loss_seals_and_reads_exact_unpublished_network_tail() { + let limits = Limits::default(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("follower-network-test"), + [15; 16], + ); + let directory = NodeDirectory::new( + layout, + Digest::from_bytes([10; 32]), + Digest::from_bytes([12; 32]), + Digest::from_bytes([13; 32]), + ); + let leader = SessionId::from_bytes([1; 16]); + let member = NodeId::from_bytes([2; 16]); + let claimant = SessionId::from_bytes([3; 16]); + let leader_key = SigningKey::from_bytes(&[21; 32]); + let member_key = SigningKey::from_bytes(&[22; 32]); + let claimant_key = SigningKey::from_bytes(&[23; 32]); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let follower_port = listener.local_addr().unwrap().port(); + let gateway_port = follower_port.checked_sub(1).unwrap(); + let endpoint = format!("https://127.0.0.1:{gateway_port}"); + let issued_at = now_ms(); + let leader_record = directory + .create( + advertisement( + NodeId::from_bytes([1; 16]), + leader, + &leader_key, + "https://127.0.0.1:8080".into(), + issued_at, + 2_000, + ), + issued_at, + ) + .await + .unwrap(); + let member_record = directory + .create( + advertisement( + member, + SessionId::from_bytes([2; 16]), + &member_key, + endpoint, + now_ms(), + 30_000, + ), + now_ms(), + ) + .await + .unwrap(); + let enrolled = directory + .recruit_log(&leader_record, 1, 1, 3, now_ms()) + .await + .unwrap(); + let root = tempfile::TempDir::new().unwrap(); + let store = FollowerStore::open( + root.path().join("follower"), + limits, + cellule_ltx::DiskBudget::new(1 << 20), + ) + .unwrap(); + let server = serve(listener, member, store, directory.clone(), member_key); + let members = HashMap::from([(member, member_record.advertisement().clone())]); + let transport = ProcessFollowerTransport::new( + leader, + leader_key.clone(), + directory.clone(), + members.clone(), + None, + ) + .unwrap(); + assert!( + transport + .append( + NodeId::from_bytes([9; 16]), + AppendRequest { + leader_session: leader, + log_epoch: 1, + frames: vec![Bytes::from_static(b"untrusted")], + covered_through: 0, + }, + ) + .await + .is_err() + ); + let expected = frame(limits); + let receipt = transport + .append( + member, + AppendRequest { + leader_session: leader, + log_epoch: 1, + frames: vec![expected.clone()], + covered_through: 0, + }, + ) + .await + .unwrap(); + assert_eq!(receipt.durable_through, 1); + let enrollment = ProcessEnrollment::new(enrolled, directory.clone(), leader); + let next = advertisement( + NodeId::from_bytes([1; 16]), + leader, + &leader_key, + "https://127.0.0.1:8080".into(), + now_ms(), + 2_000, + ); + let (activated, refreshed) = tokio::join!(enrollment.activate(1), enrollment.refresh(next)); + activated.unwrap(); + refreshed.unwrap(); + assert!( + directory + .load(leader, now_ms()) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .unwrap() + .active() + ); + assert!( + transport + .append( + member, + AppendRequest { + leader_session: leader, + log_epoch: 2, + frames: vec![expected.clone()], + covered_through: 0, + } + ) + .await + .is_err() + ); + assert!( + transport + .retire( + member, + RetireRequest { + leader_session: leader, + log_epoch: 1, + covered_through: 1, + } + ) + .await + .is_err() + ); + let claimant_record = directory + .create( + advertisement( + NodeId::from_bytes([3; 16]), + claimant, + &claimant_key, + "https://127.0.0.1:8082".into(), + now_ms(), + 30_000, + ), + now_ms(), + ) + .await + .unwrap(); + let _ = claimant_record; + tokio::time::sleep(Duration::from_millis(2_100)).await; + let fenced = directory + .claim_expired(leader, claimant, now_ms()) + .await + .unwrap(); + assert_eq!(fenced.log().unwrap().tiered_through(), 0); + let recovery_transport: Arc = Arc::new( + ProcessFollowerTransport::new(claimant, claimant_key, directory, members, None) + .unwrap(), + ); + let recovery = NodeLogRecovery::from_fenced(recovery_transport, &fenced, limits).unwrap(); + let sealed = recovery.ensure_sealed().await.unwrap(); + assert_eq!(sealed.frame_count(), 1); + assert_eq!(sealed.frames[0].encoded(), &expected); + server.abort(); + let _ = server.await; + } } diff --git a/crates/cellule-app/tests/process_node.rs b/crates/cellule-app/tests/process_node.rs index 9d0c698..c280c09 100644 --- a/crates/cellule-app/tests/process_node.rs +++ b/crates/cellule-app/tests/process_node.rs @@ -121,6 +121,7 @@ pub(super) async fn start_configured( follower_signer, lease.clone(), host.runtime().telemetry_handle(), + Arc::clone(&durability), )); let configuration = NodeDurabilitySupervisorConfig::new( ApplicationId::from_bytes([82; 16]), @@ -194,7 +195,7 @@ pub(super) async fn start_configured( Arc::new(cellule_runtime::peer::PeerSigner::new( node_session(node), host.application().registry().release_digest(), - SigningKey::from_bytes(&[93; 32]), + signing_key(node), )), cellule_runtime::peer::PeerPrincipal { issuer: "reference-runtime".into(), From a77b6c4cfe19f5cf40ac2b47f36b906d1094bdd1 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 20:31:14 -0700 Subject: [PATCH 11/14] Record follower authority phases and capacity variability --- .../performance/2026-09-29-write-capacity.md | 44 ++++++++++++++++ crates/cellule-app/qualification/entities.py | 38 +++++++++++--- .../qualification/test_entities.py | 21 +++++++- .../tests/entities/process/observation.rs | 7 +++ .../cellule-app/tests/performance_fixture.rs | 12 +++++ crates/cellule-app/tests/process_follower.rs | 50 +++++++++++++++++-- crates/cellule-app/tests/process_node.rs | 7 ++- plans/002-write-throughput-bottleneck.md | 15 +++++- plans/004-follower-enabled-capacity-lane.md | 12 +++-- plans/README.md | 4 +- 10 files changed, 188 insertions(+), 22 deletions(-) diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index e74ddbb..204767f 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -279,6 +279,50 @@ per-node keys. The next revision corrects that mismatch and adds network append duration evidence; this capacity result remains valid for its pinned binary and must be followed by a clean full rerun. +## Clean follower-proof rerun and variability + +[CI run 36663653146](https://github.com/crabbuild/cellule/actions/runs/36663653146) +passed the workspace smoke and both three-repeat capacity jobs. The +test-merge source was `74933c331e477c36f01b0fb508fa555bf897d19f`, and both +capacity lanes used release binary SHA-256 +`e4e48c7c50c8dc5a1a2c1f542fb3df6d38517e93fbc8926476c2024dcaf9c5bd`. +The source and binary were +pinned within the run; each repeat used a fresh provider volume and the same +12-Cell schedule. Every acknowledged write passed readback, every final root +covered its receipts, and no node reported CPU throttling. + +| Shape | Object fully served offered rate, repeats 1–3 | Follower fully served offered rate, repeats 1–3 | +| --- | --- | --- | +| Uniform writes | 2, 4, 4 actions/node/s | 16, 4, 4 actions/node/s | +| One hot writable Cell | 4, 4, 16 | 4, 4, 2 | +| Skewed, 20% writes | 4, 4, 4 | 2, 4, 2 | + +Most first-overload points in both lanes missed only 1–20 scheduled arrivals, +including at 4 actions/node/s. They cannot identify a stable storage throughput +limit. At the fully served hot point, object action p95 ranged from 50.42 to +147.86 ms; follower action p95 ranged from 58.33 to 155.84 ms. The new +end-to-end member append measurement had owner-0 p95 of 17.38–141.02 ms at +the follower hot fully served points. It includes member resolution, TCP, +authority lookup, and follower fsync; it does not yet attribute those phases. +The follower owner used 13.14–14.35 object-provider requests per acknowledged +hot write versus 11.12–11.69 in the object lane. This is consistent with +extra authority reads in the signed append path, but the current counters do +not prove which read or provider wait caused the tail. The first run's hot +latency gain is therefore an observed result on that pinned binary, not a +repeatable improvement across runners. A controlled A/A baseline and phase +split are needed before optimizing this transport path. + +Follower `verification.json` SHA-256 digests are +`697bb3e5f88d90a9cf6fc802ebf025a643a3cbf479b3254838e5410ada557fb7`, +`f97b12e31794e9767c48e713010f1087c684a025ee44625272b1afb0bf0362c7`, +and `28c5fcfb130f23a2f4b2842804fd64dc99a62f46f9a5225677dd8bb762e462c6`. +Object digests are +`264f2b65f81fff66ba8d87b246b551fb2840bb8859bf74aa9de415c4a3afee7f`, +`cd37d078b95f4b870db37bb22cecd455b196c29b044b5b9bfc24c078f72add6c`, +and `93c6b119f3e76a0cf1a2319703c9f0956cb2e28970d4d28b84af2bf918b521b4`. +Raw evidence and provider image digests are under +`$HOME/Workspace/crabbuild-target/cellule-capacity-5ca5/ci-run-36663653146`. + After preparing a fresh source/binary snapshot with the Compose qualification guide, run each repeat with a new state directory and Compose project: diff --git a/crates/cellule-app/qualification/entities.py b/crates/cellule-app/qualification/entities.py index 835c46b..6b3a578 100644 --- a/crates/cellule-app/qualification/entities.py +++ b/crates/cellule-app/qualification/entities.py @@ -63,6 +63,7 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic costs = rows(control / f"node-{node}-publication-costs.tsv") appends = rows(control / f"node-{node}-follower-appends.tsv") network = rows(control / f"node-{node}-follower-network.tsv") + log_events = rows(control / f"node-{node}-node-log-events.tsv") assert responses, f"node {node}: missing command response evidence" assert executions, f"node {node}: missing command execution evidence" assert publications, f"node {node}: missing publication evidence" @@ -107,6 +108,13 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic assert int(row["at_ms"]) > 0 assert int(row["bytes"]) >= 0 and int(row["duration_us"]) >= 0 assert row["acknowledged"] in {"true", "false"} + last_covered = 0 + for row in log_events: + assert int(row["at_ms"]) > 0 and int(row["epoch"]) > 0 + assert row["phase"] in {"enrolled", "active", "coverage", "closed"} + covered = int(row["covered_through"]) + assert covered >= last_covered, "node-log coverage regressed" + last_covered = covered for window in windows: if node >= window["nodes"]: continue @@ -119,6 +127,8 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic selected_costs = [row for row in costs if start <= int(row["at_ms"]) <= end] selected_appends = [row for row in appends if start <= int(row["at_ms"]) <= end] selected_network = [row for row in network if start <= int(row["at_ms"]) <= end] + covered_before_end = [int(row["covered_through"]) for row in log_events + if int(row["at_ms"]) <= end] response_sources = {source: sum(row["source"] == source for row in selected_responses) for source in sorted(sources)} window.setdefault("node_durability", {})[node] = dict( @@ -151,7 +161,8 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic follower_append_failures=sum(row["acknowledged"] == "false" for row in selected_appends), follower_append_bytes=sum(int(row["bytes"]) for row in selected_appends), follower_network_latency=distribution([int(row["duration_us"]) for row in selected_network]), - follower_network_bytes=sum(int(row["bytes"]) for row in selected_network)) + follower_network_bytes=sum(int(row["bytes"]) for row in selected_network), + node_log_covered_through=max(covered_before_end, default=0)) return dict(response_sources={source: sum(row["source"] == source for row in responses) for source in sorted(sources)}, response_latency=distribution([int(row["response_us"]) for row in responses]), @@ -166,6 +177,9 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic follower_append_bytes=sum(int(row["bytes"]) for row in appends), acknowledged_network_appends=sum(row["acknowledged"] == "true" for row in network), follower_network_latency=distribution([int(row["duration_us"]) for row in network]), + node_log_phases=dict(Counter(row["phase"] for row in log_events)), + node_log_epochs=sorted({int(row["epoch"]) for row in log_events}), + node_log_covered_through=last_covered, completed_publications=sum(row["succeeded"] == "true" for row in publications), failed_publications=sum(row["succeeded"] == "false" for row in publications)) @@ -294,10 +308,25 @@ def verify_follower_proof(resources: dict) -> dict: assert fleet_proofs > 0, "follower lane returned no follower-proof responses" assert acknowledged_appends > 0, "missing acknowledged follower append evidence" assert network_appends > 0, "missing network follower append evidence" + for resource in resources.values(): + phases = resource["durability"]["node_log_phases"] + assert phases.get("enrolled", 0) == phases.get("active", 0) == phases.get("closed", 0) == 1, \ + "follower generation did not enroll, activate, and close exactly once" + assert resource["durability"]["node_log_epochs"] == [1], "follower epoch changed" return dict(follower_proof_responses=fleet_proofs, follower_appends=acknowledged_appends, network_follower_appends=network_appends) +def verify_root_coverage(roots: list[dict], positions: dict[int, list[int]], + identity: dict[int, tuple], cells: int) -> None: + assert [int(row["entity"]) for row in roots] == list(range(cells)) + for row in roots: + entity = int(row["entity"]) + assert (row["cell"], row["owner"], row["epoch"], row["incarnation"]) == identity[entity] + assert positions[entity], f"Cell {entity} received no acknowledged writes" + assert int(row["root_sequence"]) >= max(positions[entity]), "published root does not cover writes" + + def verify_entities(control: Path, capacity: bool = False, follower: bool = False) -> dict: assert not follower or capacity stages = (3,) if capacity else STAGES @@ -324,12 +353,7 @@ def verify_entities(control: Path, capacity: bool = False, follower: bool = Fals for rate, concurrency in POINTS: windows.append(verify_window(control, nodes, shape, rate, concurrency, len(windows), positions)) roots = rows(control / f"{evidence_prefix}-roots-{nodes}.tsv") - assert [int(row["entity"]) for row in roots] == list(range(nodes * CELLS_PER_NODE)) - for row in roots: - entity = int(row["entity"]) - assert (row["cell"], row["owner"], row["epoch"], row["incarnation"]) == identity[entity] - assert positions[entity], f"Cell {entity} received no acknowledged writes" - assert int(row["root_sequence"]) >= max(positions[entity]), "published root does not cover writes" + verify_root_coverage(roots, positions, identity, nodes * CELLS_PER_NODE) for node in range(nodes): assert sum(window["acknowledged_writes_by_node"][node] for window in windows if window["nodes"] == nodes) > 0 resources = {} diff --git a/crates/cellule-app/qualification/test_entities.py b/crates/cellule-app/qualification/test_entities.py index 0773faf..5e21b57 100644 --- a/crates/cellule-app/qualification/test_entities.py +++ b/crates/cellule-app/qualification/test_entities.py @@ -6,7 +6,7 @@ import unittest from unittest.mock import patch -from entities import destination, verify_capacity_windows, verify_follower_proof, verify_object_operations, verify_timing_evidence, verify_window +from entities import destination, verify_capacity_windows, verify_follower_proof, verify_object_operations, verify_root_coverage, verify_timing_evidence, verify_window class EntityWindowEvidence(unittest.TestCase): @@ -101,6 +101,8 @@ def setUp(self): "at_ms\tacknowledged\tbytes\n") (self.root / "node-0-follower-network.tsv").write_text( "at_ms\tacknowledged\tbytes\tduration_us\n") + (self.root / "node-0-node-log-events.tsv").write_text( + "at_ms\tepoch\tphase\tcovered_through\n") self.windows = [dict(nodes=3, started_ms=100000, ended_ms=110000, elapsed_us=10_000_000)] def test_response_winner_and_later_publication_are_separate(self): @@ -226,7 +228,9 @@ class FollowerProofEvidence(unittest.TestCase): def test_follower_lane_requires_proof_and_acknowledged_append(self): resources = {0: dict(durability=dict(response_sources=dict(Fleet=0), acknowledged_follower_appends=1, - acknowledged_network_appends=1))} + acknowledged_network_appends=1, + node_log_phases=dict(enrolled=1, active=1, closed=1), + node_log_epochs=[1]))} with self.assertRaisesRegex(AssertionError, "no follower-proof responses"): verify_follower_proof(resources) resources[0]["durability"]["response_sources"]["Fleet"] = 1 @@ -241,6 +245,19 @@ def test_follower_lane_requires_proof_and_acknowledged_append(self): self.assertEqual(verify_follower_proof(resources), dict(follower_proof_responses=1, follower_appends=2, network_follower_appends=2)) + resources[0]["durability"]["node_log_phases"]["active"] = 0 + with self.assertRaisesRegex(AssertionError, "did not enroll, activate, and close"): + verify_follower_proof(resources) + + def test_root_drain_requires_every_acknowledged_sequence(self): + identity = {0: ("cell", "0", "1", "incarnation")} + positions = {0: [1, 2]} + roots = [dict(entity="0", cell="cell", owner="0", epoch="1", + incarnation="incarnation", root_sequence="1")] + with self.assertRaisesRegex(AssertionError, "published root does not cover writes"): + verify_root_coverage(roots, positions, identity, 1) + roots[0]["root_sequence"] = "2" + verify_root_coverage(roots, positions, identity, 1) if __name__ == "__main__": diff --git a/crates/cellule-app/tests/entities/process/observation.rs b/crates/cellule-app/tests/entities/process/observation.rs index 43c99e3..8767ca2 100644 --- a/crates/cellule-app/tests/entities/process/observation.rs +++ b/crates/cellule-app/tests/entities/process/observation.rs @@ -52,6 +52,7 @@ pub(super) struct NodeObservations { publication_costs: BufWriter, follower_appends: BufWriter, follower_network: BufWriter, + node_log_events: BufWriter, } impl NodeObservations { @@ -74,6 +75,7 @@ impl NodeObservations { publication_costs: create("publication-costs"), follower_appends: create("follower-appends"), follower_network: create("follower-network"), + node_log_events: create("node-log-events"), } } @@ -265,6 +267,10 @@ impl NodeObservations { ) .unwrap(); } + writeln!(self.node_log_events, "at_ms\tepoch\tphase\tcovered_through").unwrap(); + for (at_ms, epoch, phase, through) in durability.node_log_events() { + writeln!(self.node_log_events, "{at_ms}\t{epoch}\t{phase}\t{through}").unwrap(); + } self.objects.flush().unwrap(); self.object_operations.flush().unwrap(); self.durability.flush().unwrap(); @@ -276,6 +282,7 @@ impl NodeObservations { self.publication_costs.flush().unwrap(); self.follower_appends.flush().unwrap(); self.follower_network.flush().unwrap(); + self.node_log_events.flush().unwrap(); } } diff --git a/crates/cellule-app/tests/performance_fixture.rs b/crates/cellule-app/tests/performance_fixture.rs index e102c3c..1ef98cb 100644 --- a/crates/cellule-app/tests/performance_fixture.rs +++ b/crates/cellule-app/tests/performance_fixture.rs @@ -155,6 +155,7 @@ pub(super) struct DurabilityRecorder { publication_costs: Mutex>, follower_appends: Mutex>, follower_network: Mutex>, + node_log_events: Mutex>, } impl DurabilityRecorder { @@ -210,6 +211,17 @@ impl DurabilityRecorder { pub(super) fn follower_network(&self) -> Vec<(i64, bool, u64, Duration)> { self.follower_network.lock().unwrap().clone() } + + pub(super) fn record_node_log_event(&self, epoch: u64, phase: &'static str, through: u64) { + self.node_log_events + .lock() + .unwrap() + .push((now_ms(), epoch, phase, through)); + } + + pub(super) fn node_log_events(&self) -> Vec<(i64, u64, &'static str, u64)> { + self.node_log_events.lock().unwrap().clone() + } } impl CellTelemetry for DurabilityRecorder { diff --git a/crates/cellule-app/tests/process_follower.rs b/crates/cellule-app/tests/process_follower.rs index 0bc73f5..753beb9 100644 --- a/crates/cellule-app/tests/process_follower.rs +++ b/crates/cellule-app/tests/process_follower.rs @@ -645,6 +645,7 @@ pub(super) struct ProcessEnrollment { pub(super) observed: Arc>, directory: NodeDirectory, session: SessionId, + observation: Option>, } impl ProcessEnrollment { @@ -652,11 +653,19 @@ impl ProcessEnrollment { observed: VersionedNodeAdvertisement, directory: NodeDirectory, session: SessionId, + observation: Option>, ) -> Self { Self { observed: Arc::new(Mutex::new(observed)), directory, session, + observation, + } + } + + fn record_log_event(&self, epoch: u64, phase: &'static str, through: u64) { + if let Some(observation) = &self.observation { + observation.record_node_log_event(epoch, phase, through); } } @@ -694,6 +703,7 @@ impl NodeLogAuthority for ProcessEnrollment { return Err(Error::Fenced); } *observed = self.directory.activate_log(&observed, now_ms()).await?; + self.record_log_event(epoch, "active", 0); Ok(()) }) } @@ -712,6 +722,7 @@ impl NodeLogAuthority for ProcessEnrollment { .directory .advance_log_coverage(&observed, through, now_ms()) .await?; + self.record_log_event(epoch, "coverage", through); Ok(()) }) } @@ -730,6 +741,7 @@ impl NodeLogAuthority for ProcessEnrollment { .directory .close_log(&observed, barrier, now_ms()) .await?; + self.record_log_event(barrier.log_epoch(), "closed", barrier.covered_through()); Ok(()) }) } @@ -798,6 +810,7 @@ impl NodeDurabilityProvider for ProcessDurabilityProvider { return Ok(None); }; *observed = next; + self.enrollment.record_log_event(1, "enrolled", 0); } let log = observed .advertisement() @@ -1028,7 +1041,7 @@ mod tests { &leader_key, "https://127.0.0.1:8080".into(), issued_at, - 2_000, + 5_000, ), issued_at, ) @@ -1097,14 +1110,43 @@ mod tests { .await .unwrap(); assert_eq!(receipt.durable_through, 1); - let enrollment = ProcessEnrollment::new(enrolled, directory.clone(), leader); + let mut interrupted = request(Operation::Append, leader, 1); + interrupted.sender = leader.as_bytes().to_vec(); + interrupted.member = member.as_bytes().to_vec(); + interrupted.frames = vec![expected.to_vec()]; + interrupted.deadline_ms = now_ms() + MAX_DEADLINE_AHEAD_MS; + let mut socket = TcpStream::connect(("127.0.0.1", follower_port)) + .await + .unwrap(); + send( + &mut socket, + &signed(interrupted.encode_to_vec(), &leader_key, REQUEST_DOMAIN), + MAX_REQUEST_BYTES, + ) + .await + .unwrap(); + drop(socket); + let retried = transport + .append( + member, + AppendRequest { + leader_session: leader, + log_epoch: 1, + frames: vec![expected.clone()], + covered_through: 0, + }, + ) + .await + .unwrap(); + assert_eq!(retried.durable_through, 1); + let enrollment = ProcessEnrollment::new(enrolled, directory.clone(), leader, None); let next = advertisement( NodeId::from_bytes([1; 16]), leader, &leader_key, "https://127.0.0.1:8080".into(), now_ms(), - 2_000, + 5_000, ); let (activated, refreshed) = tokio::join!(enrollment.activate(1), enrollment.refresh(next)); activated.unwrap(); @@ -1162,7 +1204,7 @@ mod tests { .await .unwrap(); let _ = claimant_record; - tokio::time::sleep(Duration::from_millis(2_100)).await; + tokio::time::sleep(Duration::from_millis(5_100)).await; let fenced = directory .claim_expired(leader, claimant, now_ms()) .await diff --git a/crates/cellule-app/tests/process_node.rs b/crates/cellule-app/tests/process_node.rs index c280c09..ee2906b 100644 --- a/crates/cellule-app/tests/process_node.rs +++ b/crates/cellule-app/tests/process_node.rs @@ -109,7 +109,12 @@ pub(super) async fn start_configured( .await .unwrap(); let lease = NodeLeaseGuard::new(now_ms(), observed.advertisement().expires_at_ms()).unwrap(); - let enrollment = ProcessEnrollment::new(observed, directory.clone(), node_session(node)); + let enrollment = ProcessEnrollment::new( + observed, + directory.clone(), + node_session(node), + follower_enabled.then(|| Arc::clone(&durability)), + ); let shutdown = CancellationToken::new(); let tasks = host .install_task_group(CancellationToken::new(), shutdown.clone()) diff --git a/plans/002-write-throughput-bottleneck.md b/plans/002-write-throughput-bottleneck.md index e773a05..cf3c72a 100644 --- a/plans/002-write-throughput-bottleneck.md +++ b/plans/002-write-throughput-bottleneck.md @@ -37,6 +37,19 @@ hot throughput was fully served at 16 and overloaded at 24 actions/node/s in each repeat, confirming the limiting phase while showing runner-dependent absolute capacity. +- **Execution update, 2026-09-30:** The separate signed-network follower lane + passed three fresh-provider repeats in each of CI runs 36662251677 and + 36663653146, with the object-only lane built from the same source and + binary in each run. The verifier checked scheduled arrivals, Fleet/Object + response winners, final root coverage, readback, resource samples, and + private-disk follower append receipts. A focused network test recovered an + exact unpublished sealed tail after owner expiry; the runtime lifecycle + test separately exercises root reconstruction and replay. Follower p95 + improved at matched hot rate on the first run, but the second run had large + provider and scheduler variance in both lanes, so a stable throughput or + latency advantage is not established. The hot object-proof limiter remains + serial publication; Plan 003 requires controlled subphase evidence before + changing it. Plan 004 tracks stronger process-kill recovery qualification. ## Why this matters @@ -204,7 +217,7 @@ measurements and a regression test for its corresponding invariant. Update - [x] A single measured bottleneck has a follow-up implementation plan, or the report says precisely why the evidence is inconclusive. - [x] Focused tests, parser tests, format, lint, boundaries, and docs checks pass. -- [ ] The follower-enabled lane runs separately with its own proof, response, +- [x] The follower-enabled lane runs separately with its own proof, response, root-drain, and recovery evidence on the same scheduled shapes. ## STOP conditions diff --git a/plans/004-follower-enabled-capacity-lane.md b/plans/004-follower-enabled-capacity-lane.md index 813f914..3d1e7fe 100644 --- a/plans/004-follower-enabled-capacity-lane.md +++ b/plans/004-follower-enabled-capacity-lane.md @@ -13,7 +13,9 @@ `NodeLogTransport` or authority enrollment adapter. - **Risk:** HIGH for transport authorization, CAS ordering, and recovery. - **Depends on:** [Plan 002](002-write-throughput-bottleneck.md). -- **Status:** TODO. +- **Status:** IN PROGRESS. The signed network lane, three-repeat capacity + evidence, and exact sealed-tail owner-loss test pass. A full process-kill + takeover of an acknowledged, unpublished capacity write is still required. The existing three-process fixture in `crates/cellule-app/tests/entities/process.rs` starts object-only nodes. @@ -91,16 +93,16 @@ durability or publication semantics as part of this measurement. ## Verification and done criteria -- [ ] Focused transport authorization, enrollment/CAS, cancellation, and +- [x] Focused transport authorization, enrollment/CAS, cancellation, and owner-loss recovery tests pass. -- [ ] Parser tests reject missing follower proof, readback, and root-drain +- [x] Parser tests reject missing follower proof, readback, and root-drain evidence; format, Clippy, boundaries, module layout, and docs checks pass. -- [ ] Three follower-enabled repeats for each shape have a fully served and +- [x] Three follower-enabled repeats for each shape have a fully served and overloaded point with the same fixed schedule and resource profile as the object-proof lane. - [ ] Every acknowledged receipt survives owner loss or has an exact authority-pinned object root; no unsafe replay or false success occurs. -- [ ] Raw logs, samples, source/binary/image digests, and checksums are kept +- [x] Raw logs, samples, source/binary/image digests, and checksums are kept outside the checkout; the dated report distinguishes response throughput from eventual publication throughput. diff --git a/plans/README.md b/plans/README.md index 3f507f7..97db4c7 100644 --- a/plans/README.md +++ b/plans/README.md @@ -7,9 +7,9 @@ and process evidence outside the checkout. | Order | Plan | Priority | Effort | Status | | --- | --- | --- | --- | --- | | 1 | [Cut forwarded routing work and prove the capacity gain](001-forwarded-routing-and-capacity.md) | P1 | L | DONE: bounded adapter hint and three paired comparisons; product ingress follow-up is external | -| 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | IN PROGRESS: hot object-proof limiter measured; uniform/skewed stability and follower lane pending | +| 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | DONE: response/publication split and both proof lanes measured; hot object-proof limiter identified, absolute rate remains runner-dependent | | 3 | [Reduce the hot Cell publication critical path](003-hot-cell-publication-critical-path.md) | P1 | M+ | TODO: serial publication measured; controlled baseline and subphase attribution required before a code change | -| 4 | [Qualify follower-proof capacity across three processes](004-follower-enabled-capacity-lane.md) | P1 | L | TODO: networked follower transport and authority enrollment fixture required | +| 4 | [Qualify follower-proof capacity across three processes](004-follower-enabled-capacity-lane.md) | P1 | L | IN PROGRESS: signed network lane and recovery test pass; full process-kill takeover remains | Status values: TODO, IN PROGRESS, DONE, BLOCKED (with reason), REJECTED (with reason). Update the row after executing the plan. From 7ba5c586f1ff2212c2ffcac007962f7c54b53dcd Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 20:37:37 -0700 Subject: [PATCH 12/14] Keep optimization plans outside repository --- .../performance/2026-09-29-write-capacity.md | 27 +- crates/cellule-peer-http/docs/routing.md | 4 +- plans/001-forwarded-routing-and-capacity.md | 223 ---------------- plans/002-write-throughput-bottleneck.md | 238 ------------------ .../003-hot-cell-publication-critical-path.md | 142 ----------- plans/004-follower-enabled-capacity-lane.md | 113 --------- plans/README.md | 27 -- 7 files changed, 16 insertions(+), 758 deletions(-) delete mode 100644 plans/001-forwarded-routing-and-capacity.md delete mode 100644 plans/002-write-throughput-bottleneck.md delete mode 100644 plans/003-hot-cell-publication-critical-path.md delete mode 100644 plans/004-follower-enabled-capacity-lane.md delete mode 100644 plans/README.md diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index 204767f..9433280 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -180,9 +180,9 @@ Owner `not_started` capacity refusals began at the 32 actions/node/s point. Together these observations identify serialized object publication as the hot Cell throughput limiter on this object-proof profile. They do not yet distinguish predecessor verification, directory work, immutable uploads, or -compaction within preparation; [Plan 003](../../../plans/003-hot-cell-publication-critical-path.md) -requires that split before code changes. Uniform and skewed thresholds varied -between repeats, so this result is specific to the hot shape. +compaction within preparation; that split is required before code changes. +Uniform and skewed thresholds varied between repeats, so this result is +specific to the hot shape. The three `verification.json` SHA-256 digests are `40e4908a169013a80fe873f4aaf0d6f355872545a72212addb0d712281724e31`, @@ -215,7 +215,7 @@ p95 37–43 ms, and root preparation p95 32–38 ms. Provider requests per acknowledged write were 11.66–11.73. This independently supports serial object publication as the hot Cell limiter, while the lower hot rate interval than run 36655966439 shows that exact throughput is sensitive to the shared -CI runner. Plan 003 requires a controlled A/A baseline before claiming a +CI runner. A controlled A/A baseline is required before claiming a code-change gain. The `verification.json` SHA-256 digests for repeats 1–3 are @@ -344,11 +344,12 @@ overload point identifies the harness admission limit until runtime and provider phase evidence demonstrates a narrower Cell limiter. The existing entity host does not enroll a follower durability lane, so this -profile cannot supply the separate follower-enabled comparison. [Plan 004](../../../plans/004-follower-enabled-capacity-lane.md) -specifies the required fixture and evidence. The added -publication observation aggregates root preparation and provider I/O; it -cannot by itself distinguish predecessor GET/HEAD, immutable PUT, and provider -wait inside that phase. Those limits prevent selecting a safe implementation +profile cannot supply the separate follower-enabled comparison. That lane +requires a networked node-log transport, authority enrollment fixture, and +recovery evidence. The added publication observation aggregates root +preparation and provider I/O; it cannot by itself distinguish predecessor +GET/HEAD, immutable PUT, and provider wait inside that phase. Those limits +prevent selecting a safe implementation change from this revision alone. The 2026-09-29 workstation did not provide an isolated provider environment. @@ -373,8 +374,8 @@ above. That binary predates the fixed-Cell capacity selector and cannot run it; create a new source snapshot and release binary for the capacity runs. This is a compile result, not an execution result. -Plan 003 requires a dedicated provider environment and a repeatable A/A -rate interval before modifying publication. It then splits root preparation -into predecessor reads, directory work, uploads, provider wait, and CAS. -Plan 004 runs the follower-enabled lane separately, with recovery proof and +A publication optimization requires a dedicated provider environment and a +repeatable A/A rate interval before modifying publication. It then splits root +preparation into predecessor reads, directory work, uploads, provider wait, +and CAS. A follower-enabled lane runs separately, with recovery proof and eventual root drain alongside response throughput. diff --git a/crates/cellule-peer-http/docs/routing.md b/crates/cellule-peer-http/docs/routing.md index caebe2d..934b6e4 100644 --- a/crates/cellule-peer-http/docs/routing.md +++ b/crates/cellule-peer-http/docs/routing.md @@ -80,5 +80,5 @@ shares the sender hint and passes a known description through `with_observed_description`. A product comparison should schedule the same local and forwarded actions across hot and many-Cell stages, retain owner-loss and receipt evidence, and count object-store requests per action. Write -throughput attribution is tracked separately by the entity workload and -[`plans/002-write-throughput-bottleneck.md`](../../../plans/002-write-throughput-bottleneck.md). +throughput attribution is tracked separately by the entity workload and the +[write-capacity report](../../cellule-app/performance/2026-09-29-write-capacity.md). diff --git a/plans/001-forwarded-routing-and-capacity.md b/plans/001-forwarded-routing-and-capacity.md deleted file mode 100644 index 384853b..0000000 --- a/plans/001-forwarded-routing-and-capacity.md +++ /dev/null @@ -1,223 +0,0 @@ -# Plan 001: Reduce forwarded owner lookup cost and prove the gain - -> **Executor:** Follow the steps in order and preserve raw evidence outside the -> checkout. Do not weaken owner fencing, peer authentication, or the durable -> response gate. If a STOP condition applies, report it before editing further. -> -> **Drift check:** `git diff --stat dc387a8..HEAD -- crates/cellule-peer-http crates/cellule-runtime/src/client` -> If a listed file changed, compare the current-state facts below with live -> code before starting. Re-plan any changed behavior. - -## Status - -- **Priority:** P1 -- **Effort:** L, split into measurement, implementation, and qualification commits -- **Risk:** MED, because a stale route or ambiguous mutation must fail safely -- **Depends on:** none -- **Category:** performance -- **Planned at:** `dc387a8`, 2026-09-29 -- **Execution update, 2026-09-29:** Adapter benchmark, bounded owner hint, - failure tests, and paired comparison completed. The application ingress - follow-up remains with the embedding application maintainer. - -## Why this matters - -The generic non-owner path may read catalog and control at ingress, then read -control and a signed node advertisement in the HTTP peer sender before the -request reaches the owner. The sender repeats its two reads on each request. -Removing those reads from a healthy warm path could lower latency and object -store load, but the benefit must be measured against the peer RTT and owner -execution cost. A route observation is only a destination hint; the receiving -owner still verifies and admits every request. - -## Current state - -- `crates/cellule-runtime/src/client/runtime.rs:53-59,76-92`: the default - `RuntimeCellTransport` resolver loads catalog and exact control before - deciding whether a handle is local. A product may supply another resolver. -- `crates/cellule-peer-http/src/lib.rs:93-128,163-195`: `send_inner` permits - two attempts inside one deadline; `owner` loads exact control and then the - signed directory record on each attempt. `lib.rs:198-225` caches pinned - HTTP clients, not owner observations. -- `crates/cellule-runtime/src/client/mod.rs:618-627` provides - `with_observed_description` so a caller with an authority description can - skip `Describe`. Do not create a second description mechanism. -- `crates/cellule-runtime/src/peer/dispatch/mod.rs:134-153` authorizes a - pre-resolved peer request; actor admission fences a stale owner. -- `crates/cellule-peer-http/src/tls.rs:214-220` uses HTTP/1.1 and an existing - idle connection pool. Change it only if connection evidence calls for it. -- `crates/cellule-app/tests/host.rs` has forwarded action probes, but its - `peer_round_trip` is a custom TCP fixture from `tests/fleet.rs`. It does - **not** exercise `PeerHttpRoundTrip` and cannot measure this adapter change. -- `crates/cellule-runtime/docs/vfs-ltx-scale-plan.md` is a historical design - record. It says an embedding product implemented a bounded owner hint; - inspect that product separately before proposing ingress edits there. - -Follow the root and nearest crate `AGENTS.md`: preserve source errors, use -no `unwrap`, `expect`, or `panic!` outside tests, and keep tests beside the -module. Product HTTP endpoints and authorization are outside Cellule. - -## Scope - -**May modify:** `crates/cellule-peer-http/src/lib.rs`, `src/tests.rs`, -`docs/routing.md`, and a test-only feature declaration in `Cargo.toml` for -the counting-store fixture. A new sibling test module may be added if the -module-layout check requires one. The executor may update `plans/README.md` status. - -**Do not modify:** runtime authority, catalog, actor, peer protobuf, LTX, -follower proof, SQLite durability, read replica policy, product routing, or -resource limits. Do not add a public cache configuration option. - -## Commands - -Use a unique target directory beneath a mounted `$HOME/Workspace/crabbuild-target`. -For this plan, the examples use `cellule-route-001`: - -| Check | Command | Expected | -| --- | --- | --- | -| Adapter tests | `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-route-001" cargo test -p cellule-peer-http --locked` | All pass. | -| Runtime peer tests | `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-route-001" cargo test -p cellule-runtime peer --locked` | All pass. | -| Format | `cargo fmt --all --check` | Exit 0. | -| Lint | `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-route-001" cargo clippy --workspace --all-targets --all-features --locked -- -D warnings` | Exit 0. | -| Layout/boundaries | `python3 scripts/check-module-layout.py` and `python3 scripts/check-boundaries.py` | Both exit 0. | -| Docs/contracts | `python3 scripts/check-doc-rust-fences.py`, `python3 scripts/check-doc-links.py`, `node crates/cellule-runtime/docs/validate.mjs` | All exit 0. | - -Run broad suites and provider/process tests in CI or an isolated snapshot. -Use a disposable store and fresh object prefix. Record revision, binary digest, -limits, workload parameters, raw samples, and failures for every comparison. - -## Steps - -### 1. Establish an adapter-specific baseline - -Create one ignored benchmark test in `cellule-peer-http/src/tests.rs`, modeled -on the existing `fixture`, `TestClients`, and local Axum response fixtures. -Set up a valid published control record and signed owner advertisement using -the existing authority/directory APIs. Use a fixture store that counts exact -control and node-record reads; do not bypass record validation. Return a -valid encoded peer reply. Measure one cold send and at least 1,000 warm sends -at concurrency 1 and 16. Record lookup time, HTTP time, end-to-end adapter -p50/p95/p99, successful requests/s, send attempts, and metadata reads per -logical request. A small non-ignored test must assert the report has all -lanes and cannot silently contain zero samples. - -Run the benchmark three times on `dc387a8` under fixed CPU and network -conditions, saving raw samples outside the checkout. Check `--list` for the -exact ignored test selector before running it, then require one executed test. -Do not present this local adapter result as product ingress or RustFS capacity. - -**Verify:** the adapter test command passes; each of three baseline reports -contains at least 1,000 warm samples in each lane, exact attempted/success -counts, and the expected nonzero control and directory reads. If sender -lookup is below 10% of adapter p95 in all three runs, STOP and report the -dominant measured phase instead of adding a cache. - -### 2. Add a bounded advisory owner observation - -In `cellule-peer-http/src/lib.rs`, share a private owner-observation map -across `PeerHttpRoundTrip` clones. Key by `CellId` only after -`scope.check_target`. Store the owner session, parsed enrolled endpoint, -pinned certificate and public key, insertion time, and signed node expiry. -Populate only after the existing successful control load, directory load, -and endpoint-match check. Cap at 4,096 entries. Expire each observation at -the earlier of insertion plus five seconds and signed advertisement expiry -minus one second. Do not cache absence or authorization errors. - -The first attempt may use a live observation. A proven not-started refusal -invalidates only the attempted session and forces the existing one -authoritative retry under the original deadline. An invalid/lost response -remains an unknown outcome and is never blindly resent. A delayed older -lookup must not replace a newer observation. Hold no cache lock across -provider I/O, TLS construction, or network send. Preserve signed bytes, -peer hop limit, receiver authorization, and actor fencing. - -**Verify:** adapter tests pass. Add a counting-store assertion that two -healthy sends to one Cell cause one sender control read and one directory -read total; an expired hint causes one new lookup; and one stale hint causes -at most one authoritative retry. - -### 3. Prove failure behavior - -Extend `cellule-peer-http/src/tests.rs` using its existing HTTP response -classification fixtures. Cover owner transfer, node expiration or -retirement, endpoint/key change, deletion, five simultaneous expired-hint -callers, 429/503 with and without `Retry-After`, a delayed old lookup, and -response loss after a mutation may have been accepted. Test cancellation -releases cache state and a later request can proceed. Verify that only a -proven not-started outcome is retried and an ambiguous mutation keeps its -unknown-outcome classification. Do not infer correctness from a cache hit. - -**Verify:** adapter and runtime peer test commands pass; instrumented tests -show no resend after unknown outcome, at most two sends on a proven -not-started retry, and no send to a stale session after invalidation. - -### 4. Compare and decide - -Repeat the exact Step 1 adapter benchmark three times on the candidate, -using the same CPU/network limits and request count. Compare cold and warm -p50/p95/p99, successful requests/s at both concurrency levels, control and -directory reads/request, attempts, 503s, and unknown outcomes. Keep raw -samples and per-run paired comparisons. - -**Verify:** warm healthy sends have zero sender control/directory reads. -Keep the cache only if warm adapter p95 improves at least 10% in two of three -paired runs **or** fully successful requests/s at concurrency 16 improves at -least 10%, with no p99 regression beyond 5%, no increase in unknown outcomes -or unexplained 503s, and no increase in metadata reads per action. These -are experiment gates, not a product SLO. If the gate fails, revert the cache -change and retain the benchmark evidence. - -### 5. Record the next end-to-end and throughput work - -Update `cellule-peer-http/docs/routing.md` with the exact benchmark command, -baseline/candidate revisions, paired results, and what the adapter benchmark -does and does not cover. Hand the embedding application maintainer two -specific checks: (1) determine whether ingress shares its bounded owner -hint with the outgoing transport and passes its observed description through -`with_observed_description`; (2) run same-image, fixed-offered-rate local -versus forwarded actions across hot-Cell and many-Cell stages, with scheduled -arrivals, owner loss during load, raw receipt verification, retries, and -object-store requests/action. Inspect that product repo before prescribing -its file edits. The adapter cache alone cannot remove entry-router reads. - -For write throughput, use the existing response-source telemetry and the -phase experiments in `crates/cellule-runtime/docs/vfs-ltx-scale-plan.md` -work packet 4. Attribute queue, SQLite, capture, follower proof, root -preparation, provider wait, and control CAS; identify the largest measured -limiter and create a separate implementation plan for it. Do not change -publication or SQLite policy under this routing plan. - -**Verify:** the routing document contains the paired adapter results, -the remaining dominant adapter phase, and the product/throughput follow-up -owner. The docs/contracts commands above pass. - -## Done criteria - -- [x] Three raw baseline and three raw candidate adapter reports exist outside - the checkout, with cold/warm and concurrency 1/16 samples. -- [x] A warm sender performs zero control/directory reads while its hint is live; miss and - invalidation retain bounded exact reads and the original deadline. -- [x] Unknown mutation outcomes are never blindly replayed; stale-session, - cancellation, authentication, and refusal tests pass. -- [x] The numerical Step 4 gate passes, or the cache is reverted and the - measured reason is documented. -- [x] Focused tests, format, lint, boundaries, layout, and docs/contracts pass. -- [x] Only in-scope source/docs files and `plans/README.md` change. - -## STOP conditions - -- Drift changed the owner-resolution or retry contract. -- The fixture cannot distinguish an ambiguous outcome from a proven - not-started refusal. -- A hint would need to grant ownership, bypass receiver authorization, - extend a signed lease, or relax durable response gating. -- Step 1 shows sender lookup is not a material adapter latency phase. -- A failing verification recurs after one focused repair attempt, or a - required edit lies outside Scope. - -## Maintenance notes - -Review cache invalidation when peer response codes or node advertisement -fields change. Keep hint expiry shorter than the signed session lease. -Connection-pool tuning, read replicas, product ingress routing, and write -publication need their own measured decisions. diff --git a/plans/002-write-throughput-bottleneck.md b/plans/002-write-throughput-bottleneck.md deleted file mode 100644 index cf3c72a..0000000 --- a/plans/002-write-throughput-bottleneck.md +++ /dev/null @@ -1,238 +0,0 @@ -# Plan 002: Measure the write-throughput limit and select one safe optimization - -> **Executor:** This is a measurement and decision plan. Do not change -> publication, SQLite, or LTX semantics before the bottleneck is demonstrated. -> Keep all raw provider evidence outside the checkout. Stop on an invariant -> failure rather than altering the qualification profile. -> -> **Drift check:** `git diff --stat dc387a8..HEAD -- crates/cellule-runtime/src/cell crates/cellule-runtime/src/publication crates/cellule-runtime/src/fleet/telemetry.rs crates/cellule-ltx/src/replica crates/cellule-app/tests crates/cellule-app/qualification` -> Re-read any changed path against the facts below before editing. - -## Status - -- **Priority:** P1 -- **Effort:** M for attribution and workload evidence; later optimization is a separate plan -- **Risk:** LOW for bounded instrumentation, HIGH for an unmeasured durability change -- **Depends on:** none; compare its results with Plan 001 before prioritizing code -- **Category:** performance -- **Planned at:** `dc387a8`, 2026-09-29 -- **Execution update, 2026-09-29:** Separate response, publication, LTX, - capture, and upload observations are implemented. A fixed 12-Cell capacity - selector and verifier are implemented and locally tested. Three isolated - object-proof repeats passed in CI runs 36649534205 and 36651191247. The - finer rate ramp and per-operation provider timing narrowed tested bounds, - but the owner response/publication gap and scheduler-late skewed arrivals - leave the limiter unresolved. Actor queue and SQL worker timing passed - integrity checks in CI run 36653277555 attempt 1, but host scheduling - outliers at 4 actions/node/s made its saturation curves inconclusive. A - same-revision rerun built but could not start because the bucket-init image - registry rate-limited the pull. Final CI run 36655966439 passed three - repeats with the same hot threshold (24 fully served, 32 overloaded actions - per node per second) and attributed hot backlog to serial object - publication ahead of a roughly 2 ms SQL worker. Plan 003 now targets that - measured limiter. Uniform and skewed thresholds and the follower-enabled - lane remain unresolved. Plan 004 specifies the required networked follower - fixture and separate capacity evidence. CI run 36659182959 passed another - three object-proof repeats after the bucket-init image registry change; - hot throughput was fully served at 16 and overloaded at 24 actions/node/s - in each repeat, confirming the limiting phase while showing runner-dependent - absolute capacity. -- **Execution update, 2026-09-30:** The separate signed-network follower lane - passed three fresh-provider repeats in each of CI runs 36662251677 and - 36663653146, with the object-only lane built from the same source and - binary in each run. The verifier checked scheduled arrivals, Fleet/Object - response winners, final root coverage, readback, resource samples, and - private-disk follower append receipts. A focused network test recovered an - exact unpublished sealed tail after owner expiry; the runtime lifecycle - test separately exercises root reconstruction and replay. Follower p95 - improved at matched hot rate on the first run, but the second run had large - provider and scheduler variance in both lanes, so a stable throughput or - latency advantage is not established. The hot object-proof limiter remains - serial publication; Plan 003 requires controlled subphase evidence before - changing it. Plan 004 tracks stronger process-kill recovery qualification. - -## Why this matters - -Removing a forwarding lookup may improve a request but cannot raise a hot -Cell's write rate if its publisher or durability proof is saturated. The -existing response telemetry distinguishes recorded, follower, and object -proofs, while root preparation and SQL/capture costs are not yet tied to the -same action and offered-load curve. This plan establishes the sustainable -rate and identifies the single phase that limits it. - -## Current state and contracts - -- `crates/cellule-runtime/src/cell/actor/requests.rs:289` owns - `prove_command`; `src/cell/actor/admission.rs:146` reports the final - `CommandResponseSource` and elapsed/confirmation time. -- `crates/cellule-runtime/src/fleet/telemetry.rs:115-177` defines the bounded - `CellTelemetry` callbacks for response source, publication cost, LTX phase, - control reads, and admission. New labels must be finite; do not use Cell IDs - or tenant strings. -- `crates/cellule-runtime/src/publication/mod.rs` serializes each Cell's - object-root preparation and authority CAS. A follower proof may release a - response before object publication finishes; completed publication time is - therefore not automatically response latency. -- `crates/cellule-ltx/src/db/mod.rs:520` has `capture_deferred`, the path used - by the runtime; a `capture()` comparison has a different sync barrier. -- `crates/cellule-ltx/src/replica/prepare.rs:448-461,549` prepares immutable - roots and uploads dependencies. The predecessor graph still checks origin - presence, and missing metadata must fail closed. -- `crates/cellule-app/tests/process_scaling.rs:392-480` schedules 300 writes - 200 ms apart across a 60-second mixed-reader window. Its Python verifier in - `qualification/scale.py` requires that schedule. It is a correctness and - 5 writes/s profile, not a maximum-throughput curve. -- `crates/cellule-runtime/docs/vfs-ltx-scale-plan.md` work packet 4 lists - outstanding phase and saturation experiments. Its dated 0.3 ms capture and - 87–139 ms root-preparation figures use different harnesses and cannot be - subtracted to infer end-to-end latency. - -Preserve one fenced writer, ordered receipts, exact-root reconstruction, and -the rule that a successful command follows object publication or a -recoverable follower-log proof. Read root and nearest crate `AGENTS.md` before -changing a crate. - -## Scope - -**May modify:** finite phase instrumentation in -`crates/cellule-runtime/src/cell/actor/requests.rs`, -`src/cell/actor/admission.rs`, `src/fleet/telemetry.rs`, and focused sibling -tests; a benchmark-only workload under `crates/cellule-app/tests/` with its -integration-suite entry; a corresponding parser/test under -`crates/cellule-app/qualification/`; and a dated report under -`crates/cellule-app/performance/`. Add other production instrumentation only -after a scope review names the exact file and callback. - -**Do not modify:** SQL sync mode, LTX format, root validation, authority CAS, -follower quorum, existing qualification schedules, resource ceilings, or -the root's expected evidence. Add a separate workload selector; do not -parameterize the existing 300-write verifier until its contract tests are -updated independently. - -## Commands - -| Check | Command | Expected | -| --- | --- | --- | -| Focused runtime tests | `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-capacity-002" cargo test -p cellule-runtime --features test-support --locked` | Pass. | -| App integration | `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-capacity-002" cargo test -p cellule-app --test integration --locked` | Non-ignored tests pass. | -| Evidence parser | `python3 -m unittest discover -s crates/cellule-app/qualification -p test_scale.py` | Pass. | -| Format and lint | `cargo fmt --all --check` and `CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-capacity-002" cargo clippy --workspace --all-targets --all-features --locked -- -D warnings` | Both pass. | -| Boundaries/docs | `python3 scripts/check-boundaries.py`, `python3 scripts/check-module-layout.py`, `python3 scripts/check-doc-rust-fences.py`, `python3 scripts/check-doc-links.py`, `node crates/cellule-runtime/docs/validate.mjs` | All pass. | - -Run broad suites, process faults, and provider runs in CI or an isolated -snapshot. Set `CARGO_TARGET_DIR` under the mounted Workspace volume with a -directory unique to the checkout. Use a disposable provider and new object -prefix for each run. - -## Steps - -### 1. Make response and background publication separately visible - -Add bounded phase observations around queue wait, SQL handler/commit, -`capture_deferred`, follower append/proof, root preparation, provider I/O, -authority CAS, and final confirmation. Each observation must carry only a -fixed phase/outcome label and timing. Tie phases to one request in an -isolated trace or raw benchmark record, without putting request or Cell IDs -in aggregate metric labels. Record separately (a) which proof released the -response and (b) when background object publication drained. A cancellation -or unknown mutation must not be counted as a successful response. - -**Verify:** focused runtime tests pass. Add a unit test that a follower-first -response has exactly one response winner and may have a later object proof; -an object-first response has one object winner; and cancellation produces no -false successful response. Compare event counts with existing committed -sequence assertions. - -### 2. Add a scheduled capacity workload without weakening existing tests - -Add a new ignored application integration workload with fixed Cell count and -recorded arrival times. Use at least three shapes: one hot writable Cell, -many evenly distributed Cells, and skewed Cells. At each shape, run increasing -offered rates until one fully served rate is followed by an overloaded rate. -Record every scheduled, started, completed, rejected, timed-out, and late -arrival; retain one stable mutation identity for any retry. Check each -acknowledged receipt with a minimum-receipt readback. Record owner/epoch, -CPU, memory, open Cells, publisher queue age, unpublished bytes, root lag, -object GET/HEAD/PUT and bytes, attempts, and p50/p95/p99 for successful and -all scheduled actions. The parser must reject missing samples, an inflated -success rate, a missing readback, and a rate mislabeled as fully served. - -Use the existing `process_scaling.rs` scheduled-arrival and -`qualification/scale.py` receipt checks as patterns, but give the new -workload its own selector and evidence filenames. Keep the 300-write mixed -reader workload and its tests unchanged. - -**Verify:** app tests and parser tests pass. A deliberately incomplete -synthetic report must be rejected. A completed report must show both a fully -served point and an explicit overloaded point for every workload shape. - -### 3. Run paired evidence and locate saturation - -In an isolated provider environment, run each shape three times with the -same revision, binary digest, node CPU/memory limits, provider identity, -Cell count, and offered-rate schedule. Run a follower-enabled and an -object-proof-only lane; do not combine their percentiles. Include a cold -activation phase and steady resident phase. Preserve raw samples and logs -outside the checkout, then write a dated summary under -`crates/cellule-app/performance/` linking their locations and checksums. - -Calculate the maximum **fully served** logical writes/s for each shape. -Classify each overloaded point as admission, SQL/queue, capture, follower, -publication, provider, or CPU/memory saturation using its phase and resource -evidence. Report published roots/s and root lag independently of response -throughput. A response released on follower proof does not establish that -the publisher can drain indefinitely. - -**Verify:** all three repeats agree on the dominant phase for each shape, -or the report explicitly says the result is inconclusive. Every successful -write has readback evidence; no run reports a supported rate at an overloaded -point. The report includes p95/p99 and raw evidence checksums. - -### 4. Write the next implementation plan for the measured limiter - -If root preparation dominates, inspect its predecessor GET/HEAD, directory -read, immutable PUT, provider wait, and CAS split before proposing a change. -If queue/SQL dominates, inspect worker occupancy and checkpoint/full-image -events. If follower proof dominates, inspect enrollment and append wait. -If the runs disagree or the provider is saturated externally, repeat the -measurement under an isolated profile instead of changing code. - -Write a new plan under `plans/` for **one** measured limiter. It must name -the exact source paths, unchanged safety invariants, a before/after workload, -and a numeric gate: improved fully served rate or p95/p99 without more -failed proofs, retry pressure, root lag, or object requests per logical write. -Do not implement that follow-up as part of Plan 002. - -**Verify:** the new plan identifies one limiting phase with supporting raw -measurements and a regression test for its corresponding invariant. Update -`plans/README.md` with the next plan and this plan's status. - -## Done criteria - -- [x] Response-winning proof and later publication are separately measured. -- [x] A new scheduled workload covers hot, uniform, and skewed Cells without - changing the existing mixed-reader qualification contract. -- [x] Three repeats per shape distinguish fully served from overloaded rates - and retain raw readback, resource, and object-store evidence. -- [x] A single measured bottleneck has a follow-up implementation plan, or - the report says precisely why the evidence is inconclusive. -- [x] Focused tests, parser tests, format, lint, boundaries, and docs checks pass. -- [x] The follower-enabled lane runs separately with its own proof, response, - root-drain, and recovery evidence on the same scheduled shapes. - -## STOP conditions - -- A successful response can no longer be tied to exactly one durable proof. -- The workload needs to weaken the existing 300-write/60-second profile or - omit scheduled arrivals to pass. -- A phase counter would block the actor or add unbounded-cardinality labels. -- Any acknowledged receipt fails readback or owner-loss recovery. -- The measured limiter cannot be distinguished from shared-host provider or - resource contention after three controlled runs. - -## Maintenance notes - -Review phase labels when the publication pipeline changes. Preserve both -response rate and eventual root-drain rate in future capacity reports. -Never treat a faster local `capture()` benchmark as evidence for the runtime's -`capture_deferred()` path. diff --git a/plans/003-hot-cell-publication-critical-path.md b/plans/003-hot-cell-publication-critical-path.md deleted file mode 100644 index 218eae2..0000000 --- a/plans/003-hot-cell-publication-critical-path.md +++ /dev/null @@ -1,142 +0,0 @@ -# Plan 003: Reduce the hot Cell's object-publication critical path - -> **Executor:** First establish a stable baseline on a runner with spare CPU -> for the driver, three one-CPU nodes, and RustFS. Do not change root format, -> authority fencing, response proof, or retention semantics to hit a rate. -> Stop if host scheduling or registry failures prevent a clean A/A baseline. - -## Status and evidence - -- **Priority:** P1 for object-proof throughput; follower-proof latency is a - separate lane. -- **Effort:** M for critical-path attribution; implementation effort depends - on the measured subphase. -- **Risk:** LOW for finite timing observations, HIGH for publication changes. -- **Depends on:** [Plan 002](002-write-throughput-bottleneck.md). -- **Status:** TODO. Three repeats identify serial object publication as the - hot Cell limiter; the dominant root-preparation subphase remains unmeasured. - -[The capacity report](../crates/cellule-app/performance/2026-09-29-write-capacity.md) -retains three raw object-proof repeats in each of two CI runs. In the finer -run, the hot owner's first overloaded window published about 61–66 roots/s; -mean publication time was 14–16 ms/root. Owner capacity refusals began at -24–32 actions/node/s, with zero end-window root lag and no node CPU -throttling. Provider PUT p95 was 6.7–7.5 ms, while capture p95 was under -1 ms. A later instrumented run observed 59–67 ms actor-queue p95 in two -hot overload windows while worker round-trip p95 was about 4 ms. Its -fully-served thresholds varied sharply, including scheduler-late arrivals at -only 4 actions/node/s. The final CI run 36655966439 found the same hot -threshold in all three repeats: 24 actions/node/s fully served and 32 -overloaded. At overload, owner 0 published 62–65 roots/s at 15–16 ms mean -publication time/root. Actor queue p95 rose to 142–160 ms while SQL worker -p95 stayed near 2 ms. Root preparation p95 was 33–37 ms of 37–43 ms -publication p95. The serialized publication path is the measured hot -throughput limiter; the exact subphase to change is not yet established. -Uniform and skewed thresholds still varied between repeats. -An independent three-repeat CI run 36659182959 again found serial object -publication ahead of a roughly 3 ms SQL worker at hot overload, but its hot -interval shifted to 16 fully served and 24 overloaded actions/node/s. The -different interval strengthens the requirement for a dedicated, controlled -A/A baseline before applying this plan's numeric improvement gate. - -## Scope and invariants - -Inspect `crates/cellule-runtime/src/cell/actor/requests.rs` and -`src/publication/mod.rs` for publication serialization and authority CAS; -`crates/cellule-ltx/src/replica/prepare.rs`, `upload.rs`, and -`directory/mod.rs` for predecessor verification, directory update, and -immutable uploads. The embedding application owns any HTTP, credentials, -follower transport, or deployment changes. - -Preserve one fenced writer; ordered committed outcomes; a successful response -only after an exact authority-pinned root or recoverable follower proof; all -required predecessor and chunk verification; and bounded retention/admission. -An optimization may not skip origin verification merely because local memory -contains a root or digest. Do not alter persisted IDs, object paths, LTX -formats, or signed peer messages. Read each crate's `AGENTS.md`, producers, -consumers, sibling implementations, and tests before editing. - -## Steps - -### 1. Establish a clean baseline - -Run the existing `--workload capacity` hot and uniform shapes at the same -revision and binary digest for three repeats on a dedicated runner with at -least two CPUs beyond the Compose limits for three nodes and the driver. -Record host CPU pressure and runnable-task delay as well as the existing -node cgroups, provider timings, response proof, root drain, readback, and -scheduled arrivals. Pre-pull pinned images once for the run or use a registry -with sufficient quota; retain exact image digests. No rate is fully served -if any arrival is late, refused, or missing. - -**Gate:** All three repeats reach the same fully served and first overloaded -rate interval for the hot shape, with no low-rate scheduler-late arrivals. -If this fails, fix the runner or harness and repeat; do not tune publication. - -### 2. Attribute one root's serial time - -Add finite, nonblocking timing observations around predecessor graph load, -directory update, immutable dependency upload, root-document upload, worker -`bind_prepared`, authority CAS, and worker `confirm_published`. Keep the -observed root sequence in raw local evidence for correlation, never a metric -label. Report overlap explicitly: do not sum parallel upload percentiles. -Measure count and latency of GET/HEAD/range/PUT per acknowledged write and -identify compaction windows separately from ordinary appends. - -**Gate:** For each of three repeats, the same subphase accounts for the -largest avoidable part of the hot Cell's serial critical path. Confirm that -queue growth begins only when offered writes exceed published roots/s. If -the dominant subphase differs by repeat, report the split and stop. - -### 3. Change only the measured subphase - -- If verified predecessor reads dominate, reuse only work that can be tied to - the exact current published root and still fail closed on missing origin - metadata. Test restart, takeover, missing predecessor, and retention races. -- If redundant immutable uploads dominate, prove which content-addressed - objects are already required by the current authority-pinned root before - skipping an upload. Invalidate any memo on failure or owner change, and - verify recovery after deleting a required object. -- If directory computation or compaction dominates, reduce that work without - changing the canonical checksum or compaction debt bound. Test byte-identical - roots and restoration across checkpoint and truncate/regrow cases. -- If authority CAS dominates, stop and write a separate authority protocol - proposal; do not omit or defer the fenced CAS. - -Keep one implementation path and no speculative tuning flags. A failed -preparation must leave only unreachable objects; a failed CAS must fence the -owner and cannot release a success response. - -### 4. Prove the change - -Run paired baseline/candidate builds on the same controlled runner, provider -image, Cell count, node limits, and rate schedule, with three alternating -repeats per build. Require at least **20% more fully served hot logical -writes/s** or **20% lower hot p95 response latency at the same offered rate**. -Require p99 to improve or stay within 5% of baseline, zero failed proofs, -zero readback failures, no higher root lag, and no increase in object requests -per logical write. Report published roots/s separately from response rate; -do not count follower-first responses as proof of publisher capacity. - -Run focused runtime/LTX tests, both application integration and parser tests, -format, all-feature check, Clippy, boundaries, module layout, documentation -gates, and the SQL/peer contract validator. Run broad and provider suites in -CI or an isolated verification snapshot using a checkout-specific target -under `$HOME/Workspace/crabbuild-target`. - -## Done criteria - -- [ ] A stable, integrity-verified A/A baseline identifies the same hot - publication subphase in three repeats. -- [ ] One change to that subphase preserves all proof and recovery invariants. -- [ ] Paired evidence meets the numeric rate or latency gate without increased - retry, root lag, failed proof, or object-request pressure. -- [ ] Raw samples, binary/image digests, and checksums are retained outside - the checkout; the dated report and runnable example are updated. - -## STOP conditions - -- A scheduler-late or provider-registry error prevents the baseline. -- A cache would hide a missing required object or a stale authority root. -- A proposed shortcut changes the root format, fenced CAS, or response proof. -- Any acknowledged receipt cannot be reconstructed exactly after owner loss. diff --git a/plans/004-follower-enabled-capacity-lane.md b/plans/004-follower-enabled-capacity-lane.md deleted file mode 100644 index 3d1e7fe..0000000 --- a/plans/004-follower-enabled-capacity-lane.md +++ /dev/null @@ -1,113 +0,0 @@ -# Plan 004: Qualify follower-proof write capacity across three processes - -> **Executor:** This finishes Plan 002's separate follower-enabled lane. Keep -> its object-proof workload, provider, Cell count, rate schedule, and proof -> rules fixed. A same-process `LocalFollowerTransport` is not evidence for -> networked follower capacity. Stop if an acknowledged follower cut cannot be -> recovered after owner loss. - -## Status and boundary - -- **Priority:** P1 for response-latency comparison. -- **Effort:** L; the application fixture currently has no networked - `NodeLogTransport` or authority enrollment adapter. -- **Risk:** HIGH for transport authorization, CAS ordering, and recovery. -- **Depends on:** [Plan 002](002-write-throughput-bottleneck.md). -- **Status:** IN PROGRESS. The signed network lane, three-repeat capacity - evidence, and exact sealed-tail owner-loss test pass. A full process-kill - takeover of an acknowledged, unpublished capacity write is still required. - -The existing three-process fixture in -`crates/cellule-app/tests/entities/process.rs` starts object-only nodes. -`crates/cellule-app/tests/process_node.rs` advertises nodes and renews leases, -but does not install follower stores or a durability provider. Runtime -`node/log_transport.rs` defines the transport contract and a deliberately -in-process test implementation; `node/directory/log.rs` owns authoritative -enrollment, activation, coverage, rotation, and recovery authorization. -Keep the network protocol and credentials in the embedding application or its -test fixture, outside `cellule-runtime` and `cellule-app` production APIs. - -## Steps - -### 1. Add a private-disk, authenticated follower endpoint - -Extend the process fixture with a bounded peer listener for append, seal, -retire, and paged tail. Give each node a `FollowerStore` on its existing -private `/scratch` volume. Authenticate the caller and exact session, member, -epoch, operation, and request deadline before touching that store. Use -`NodeDirectory::authorize_log_append`, `authorize_log_retire`, and -`authorize_log_recovery` as appropriate; make frame and response limits -explicit. The client implements `NodeLogTransport` and connects to the -advertised peer endpoint. Exercise refusal of wrong member, stale epoch, -unauthorized retire, oversized batch, and expired deadline in focused fixture -tests. Never treat an HTTP/TCP acknowledgment as follower fsync unless the -store returned its authenticated receipt. - -### 2. Enroll one host-owned durability generation - -In `tests/process_node.rs`, install the follower store and a -`NodeDurabilityProvider` before `host.start()`. The provider recruits a -complete ensemble through `NodeDirectory::try_recruit_log`, then builds a -`NodeDurabilityConfig` for the exact host session, node, epoch, members, -transport, lease, limits, and telemetry. Serialize its activation, coverage, -rotation, withdrawal, and heartbeat refresh against the same versioned -advertisement; reconcile ambiguous CAS only when the exact session and epoch -still match. Keep the host task group responsible for shutdown and drain. - -Add a readiness marker only after all three nodes are live, follower stores -are listening, and each owner has an enrolled generation. Verify the first -follower-proof response is preceded by all-member fsync and authoritative -activation. Test clean rotation, rejected follower append, owner loss before -object publication, sealed-tail replay, and byte-identical root recovery. - -### 3. Run the same scheduled shapes in an isolated lane - -Add a `capacity-follower` selector to -`crates/cellule-app/tests/entities/process/driver.rs` and -`qualification/scale.py`. Reuse the 12-Cell hot, uniform, and skewed schedule, -arrival accounting, minimum-receipt readback, provider counters, and resource -sampling from the object-proof lane. Record response winner, follower append -bytes and latency, node-log epoch and covered sequence, publication drain, -root lag, retries, and p50/p95/p99 for successful and all scheduled actions. -The parser rejects missing proof records, false fully-served labels, -unrecovered acknowledged writes, or publisher backlog at the end of a drain -window. Keep the existing 300-write qualification profile unchanged. - -Add a separate CI job in `.github/workflows/write-capacity.yml` with three -fresh repeats, pinned source/binary/provider digests, fixed resource limits, -and artifact retention. Pre-pull pinned images for the run and retain their -digests so registry quota cannot be mistaken for workload saturation. Compare -the two lanes only at equal offered rates and report sustainable response -rate and published roots/s separately. Run a cold activation and a steady -resident window in each repeat. - -### 4. Publish decision evidence - -Update `crates/cellule-app/performance/2026-09-29-write-capacity.md` with -per-shape bounds, response-source mix, p95/p99, root drain, provider requests -per logical write, recovery outcome, and checksums of raw external evidence. -Classify each overload using the phase and resource record. If the follower -lane releases responses faster but roots cannot drain, report the stable -response bound separately from publisher capacity. Do not change runtime -durability or publication semantics as part of this measurement. - -## Verification and done criteria - -- [x] Focused transport authorization, enrollment/CAS, cancellation, and - owner-loss recovery tests pass. -- [x] Parser tests reject missing follower proof, readback, and root-drain - evidence; format, Clippy, boundaries, module layout, and docs checks pass. -- [x] Three follower-enabled repeats for each shape have a fully served and - overloaded point with the same fixed schedule and resource profile as - the object-proof lane. -- [ ] Every acknowledged receipt survives owner loss or has an exact - authority-pinned object root; no unsafe replay or false success occurs. -- [x] Raw logs, samples, source/binary/image digests, and checksums are kept - outside the checkout; the dated report distinguishes response throughput - from eventual publication throughput. - -Run process and provider checks in CI or an isolated snapshot, with -`CARGO_TARGET_DIR` under `$HOME/Workspace/crabbuild-target` for each checkout. -The STOP conditions are any unauthorized follower operation, a successful -response without its durable proof, a failed exact recovery, or a changed -object-proof qualification profile. diff --git a/plans/README.md b/plans/README.md deleted file mode 100644 index 97db4c7..0000000 --- a/plans/README.md +++ /dev/null @@ -1,27 +0,0 @@ -# Executable performance plans - -These plans were prepared against Cellule commit `dc387a8` on 2026-09-29. -Read the whole plan and its STOP conditions before execution. Keep raw provider -and process evidence outside the checkout. - -| Order | Plan | Priority | Effort | Status | -| --- | --- | --- | --- | --- | -| 1 | [Cut forwarded routing work and prove the capacity gain](001-forwarded-routing-and-capacity.md) | P1 | L | DONE: bounded adapter hint and three paired comparisons; product ingress follow-up is external | -| 2 | [Measure the write-throughput limit](002-write-throughput-bottleneck.md) | P1 | M | DONE: response/publication split and both proof lanes measured; hot object-proof limiter identified, absolute rate remains runner-dependent | -| 3 | [Reduce the hot Cell publication critical path](003-hot-cell-publication-critical-path.md) | P1 | M+ | TODO: serial publication measured; controlled baseline and subphase attribution required before a code change | -| 4 | [Qualify follower-proof capacity across three processes](004-follower-enabled-capacity-lane.md) | P1 | L | IN PROGRESS: signed network lane and recovery test pass; full process-kill takeover remains | - -Status values: TODO, IN PROGRESS, DONE, BLOCKED (with reason), REJECTED (with -reason). Update the row after executing the plan. - -## Dependency and scope notes - -Plan 001 begins with measurement and keeps an explicit stop gate before adding -an owner hint. It covers the optional Cellule peer HTTP adapter. The embedding -application owns ingress routing; its integration work requires its own repo. -Plan 002 is independent of Plan 001 and measures the hot-Cell and fleet-wide -write limit. Plan 003 narrows the next target to serial object publication -and requires a controlled baseline and exact subphase attribution before -changing code or durability policy. -Plan 004 supplies the separate follower-enabled lane required by Plan 002; -it does not change the object-proof baseline or the hot publication target. From abe14763f4dab8c17ca98df75821add73621d1f8 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 21:00:03 -0700 Subject: [PATCH 13/14] Verify follower coverage across lifecycle markers --- crates/cellule-app/PERFORMANCE.md | 1 + .../performance/2026-09-29-write-capacity.md | 23 +++++++++---------- crates/cellule-app/qualification/entities.py | 9 +++++--- .../qualification/test_entities.py | 17 ++++++++++++++ 4 files changed, 35 insertions(+), 15 deletions(-) diff --git a/crates/cellule-app/PERFORMANCE.md b/crates/cellule-app/PERFORMANCE.md index b41d310..ccd736a 100644 --- a/crates/cellule-app/PERFORMANCE.md +++ b/crates/cellule-app/PERFORMANCE.md @@ -14,6 +14,7 @@ and [qualification runner](qualification/run.sh) for a fresh run. | Three-process RustFS | [`qualification/run.sh`](qualification/run.sh) | Local and forwarded gateway calls, receipts, drained sessions. | | Entity fleet and scaling | [`qualification/entities.py`](qualification/entities.py), [`scale.py`](qualification/scale.py) | Isolated Cell ledgers and bounded traffic. | | Fixed 12-Cell write capacity | [`qualification/scale.py`](qualification/scale.py) with `--workload capacity` | Hot, uniform, and skewed offered-rate ramps with receipt and overload checks. | +| Fixed 12-Cell follower-proof capacity | [`qualification/scale.py`](qualification/scale.py) with `--workload capacity-follower` | Networked follower proofs, offered-rate ramps, and final root coverage. | | Reader and rollout variants | [Application integration suite](tests/integration.rs) | Selection, replacement, and recovered receipts. | ```mermaid diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index 9433280..51be463 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -39,11 +39,11 @@ the last fully served logical write rate separately from the first overloaded rate. The first and second isolated object-proof results are summarized below. The dedicated `Cell write capacity qualification` GitHub Actions workflow -builds one release binary and runs three object-proof repeats on a fresh -runner. Each repeat has its own Compose project, RustFS volume, object prefix, -and evidence directory. It uploads raw samples, logs, the binary digest, and -provider details even if a repeat fails. Its output requires review before a -capacity claim; the follower-enabled comparison remains separate. +builds release binaries for separate object-proof and follower-proof jobs. +Each job runs three repeats; each repeat has its own Compose project, RustFS +volume, object prefix, and evidence directory. The workflow uploads raw +samples, logs, binary digests, and provider details even if a repeat fails. +Its output requires review before a capacity claim. ## First isolated object-proof result @@ -343,14 +343,13 @@ precision. A scheduler-late or client-full overload point identifies the harness admission limit until runtime and provider phase evidence demonstrates a narrower Cell limiter. -The existing entity host does not enroll a follower durability lane, so this -profile cannot supply the separate follower-enabled comparison. That lane -requires a networked node-log transport, authority enrollment fixture, and -recovery evidence. The added publication observation aggregates root -preparation and provider I/O; it cannot by itself distinguish predecessor +The original object-only profile could not supply a follower-enabled +comparison. The later follower-proof lane above adds a networked node-log +transport and authority enrollment; a separate process test covers owner-loss +recovery. The added publication observation aggregates root preparation and +provider I/O; it cannot by itself distinguish predecessor GET/HEAD, immutable PUT, and provider wait inside that phase. Those limits -prevent selecting a safe implementation -change from this revision alone. +prevent selecting a safe publication change from this evidence alone. The 2026-09-29 workstation did not provide an isolated provider environment. The Colima VM had multiple unrelated active RustFS workloads, several above diff --git a/crates/cellule-app/qualification/entities.py b/crates/cellule-app/qualification/entities.py index 6b3a578..ccec905 100644 --- a/crates/cellule-app/qualification/entities.py +++ b/crates/cellule-app/qualification/entities.py @@ -113,8 +113,11 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic assert int(row["at_ms"]) > 0 and int(row["epoch"]) > 0 assert row["phase"] in {"enrolled", "active", "coverage", "closed"} covered = int(row["covered_through"]) - assert covered >= last_covered, "node-log coverage regressed" - last_covered = covered + if row["phase"] in {"enrolled", "active"}: + assert covered == 0, "node-log lifecycle marker has coverage" + else: + assert covered >= last_covered, "node-log coverage regressed" + last_covered = covered for window in windows: if node >= window["nodes"]: continue @@ -128,7 +131,7 @@ def verify_timing_evidence(control: Path, node: int, windows: list[dict]) -> dic selected_appends = [row for row in appends if start <= int(row["at_ms"]) <= end] selected_network = [row for row in network if start <= int(row["at_ms"]) <= end] covered_before_end = [int(row["covered_through"]) for row in log_events - if int(row["at_ms"]) <= end] + if row["phase"] in {"coverage", "closed"} and int(row["at_ms"]) <= end] response_sources = {source: sum(row["source"] == source for row in selected_responses) for source in sorted(sources)} window.setdefault("node_durability", {})[node] = dict( diff --git a/crates/cellule-app/qualification/test_entities.py b/crates/cellule-app/qualification/test_entities.py index 5e21b57..f9b4290 100644 --- a/crates/cellule-app/qualification/test_entities.py +++ b/crates/cellule-app/qualification/test_entities.py @@ -131,6 +131,23 @@ def test_duplicate_publication_is_rejected(self): with self.assertRaisesRegex(AssertionError, "duplicate publication"): verify_timing_evidence(self.root, 0, self.windows) + def test_active_marker_after_coverage_does_not_reset_covered_sequence(self): + path = self.root / "node-0-node-log-events.tsv" + path.write_text("at_ms\tepoch\tphase\tcovered_through\n" + "100001\t1\tenrolled\t0\n" + "100002\t1\tcoverage\t1\n" + "100003\t1\tactive\t0\n" + "100004\t1\tcoverage\t2\n" + "100005\t1\tclosed\t2\n") + report = verify_timing_evidence(self.root, 0, self.windows) + self.assertEqual(report["node_log_covered_through"], 2) + self.assertEqual(self.windows[0]["node_durability"][0]["node_log_covered_through"], 2) + + path.write_text(path.read_text().replace("100004\t1\tcoverage\t2", + "100004\t1\tcoverage\t0")) + with self.assertRaisesRegex(AssertionError, "node-log coverage regressed"): + verify_timing_evidence(self.root, 0, self.windows) + class CapacityScheduleEvidence(unittest.TestCase): def setUp(self): From 061239a1d47866f0548219d407f5eb79c12d39ae Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 29 Sep 2026 21:57:31 -0700 Subject: [PATCH 14/14] Skip local metadata lookup for forwarded Cell calls --- crates/cellule-runtime/Cargo.toml | 1 + .../cellule-runtime/docs/overview-detailed.md | 12 ++- .../cellule-runtime/src/cell/actor/acquire.rs | 21 ++++ crates/cellule-runtime/src/client/mod.rs | 15 ++- crates/cellule-runtime/src/client/runtime.rs | 30 ++++-- .../cellule-runtime/tests/protocol/client.rs | 5 +- .../tests/protocol/client/typed.rs | 96 +++++++++++++++++++ 7 files changed, 160 insertions(+), 20 deletions(-) diff --git a/crates/cellule-runtime/Cargo.toml b/crates/cellule-runtime/Cargo.toml index 2ebce6e..1fe0f70 100644 --- a/crates/cellule-runtime/Cargo.toml +++ b/crates/cellule-runtime/Cargo.toml @@ -43,5 +43,6 @@ protoc-bin-vendored = "3" [dev-dependencies] async-trait.workspace = true +cellule-store = { workspace = true, features = ["test-support"] } proptest = "1" tokio = { workspace = true, features = ["macros", "rt-multi-thread"] } diff --git a/crates/cellule-runtime/docs/overview-detailed.md b/crates/cellule-runtime/docs/overview-detailed.md index 50dd4bb..e24199d 100644 --- a/crates/cellule-runtime/docs/overview-detailed.md +++ b/crates/cellule-runtime/docs/overview-detailed.md @@ -38,11 +38,13 @@ The public HTTP server owns authentication and repository policy. The runtime ow The direct green route is local execution. The orange route is the single authenticated peer hop when another node owns the Cell. Both converge on the same registry, actor, SQLite, and LTX publication path. `CellClient::local_runtime` resolves a newly admitted local Cell through its -catalog and owner record for each call. It serves embedded, single-node routing; -`CellClient::runtime_with_peer` uses the same local path when this node owns the -Cell and an authenticated peer round trip when another node owns it. The product -server supplies the peer transport and owner lookup; neither constructor -acquires an idle Cell. +catalog and owner record for each call, reading those independent records in +parallel. It serves embedded, single-node routing; +`CellClient::runtime_with_peer` first asks the local actor whether it has a +dispatchable handle. A local hit verifies catalog and owner authority as before; +a miss delegates without those metadata reads to an authenticated peer round +trip. The product server supplies the peer transport and owner lookup; neither +constructor acquires an idle Cell. The dependency direction follows the same boundary: diff --git a/crates/cellule-runtime/src/cell/actor/acquire.rs b/crates/cellule-runtime/src/cell/actor/acquire.rs index b8b3c79..16c13bc 100644 --- a/crates/cellule-runtime/src/cell/actor/acquire.rs +++ b/crates/cellule-runtime/src/cell/actor/acquire.rs @@ -7,6 +7,27 @@ use super::*; impl CellRuntime { + /// Checks actor-owned admission before a peer route reads Cell metadata. + /// + /// A miss only means this process has no currently dispatchable handle; + /// the peer transport remains responsible for resolving remote authority. + pub(crate) async fn has_local_owner(&self, cell: CellId) -> crate::Result { + self.ensure_running()?; + let (reply, response) = oneshot::channel(); + self.inner + .sender + .send(Message::Lookup { + cell, + require_resident: false, + reply, + }) + .await + .map_err(|_| Error::RuntimeClosed)?; + let local = response.await.map_err(|_| Error::RuntimeClosed)?; + self.ensure_running()?; + Ok(local.is_some()) + } + /// Resolves an active local owner without exposing the dispatcher's Cell map. pub async fn local_handle( &self, diff --git a/crates/cellule-runtime/src/client/mod.rs b/crates/cellule-runtime/src/client/mod.rs index 3475164..d19ecf9 100644 --- a/crates/cellule-runtime/src/client/mod.rs +++ b/crates/cellule-runtime/src/client/mod.rs @@ -724,9 +724,9 @@ impl CellClient { /// Routes a target to its current local owner or an authenticated peer. /// - /// Every invocation rechecks catalog and authority state. The peer round - /// trip must resolve the current remote owner and verify its enrollment; - /// this constructor does not acquire an idle Cell. + /// A local actor hit rechecks catalog and authority state. A local miss + /// delegates without those reads; the peer round trip resolves the remote + /// owner and verifies enrollment. This does not acquire an idle Cell. #[must_use] pub fn runtime_with_peer( registry: Arc, @@ -736,8 +736,13 @@ impl CellClient { principal: crate::peer::PeerPrincipal, round_trip: Arc, ) -> Self { - Self::peer(registry, signer, principal, round_trip) - .with_local_resolver(Arc::new(RuntimeLocalResolver { runtime, layout })) + Self::peer(registry, signer, principal, round_trip).with_local_resolver(Arc::new( + RuntimeLocalResolver { + runtime, + layout, + remote_on_miss: true, + }, + )) } /// Resolves a local owner before delegating to this client's transport. diff --git a/crates/cellule-runtime/src/client/runtime.rs b/crates/cellule-runtime/src/client/runtime.rs index 788c677..5605d4b 100644 --- a/crates/cellule-runtime/src/client/runtime.rs +++ b/crates/cellule-runtime/src/client/runtime.rs @@ -33,7 +33,11 @@ impl RuntimeCellTransport { ) -> Self { Self { registry, - resolver: Arc::new(RuntimeLocalResolver { runtime, layout }), + resolver: Arc::new(RuntimeLocalResolver { + runtime, + layout, + remote_on_miss: false, + }), remote: None, } } @@ -71,6 +75,7 @@ impl RuntimeCellTransport { pub(super) struct RuntimeLocalResolver { pub(super) runtime: CellRuntime, pub(super) layout: CellStorageLayout, + pub(super) remote_on_miss: bool, } impl LocalCellResolver for RuntimeLocalResolver { @@ -80,14 +85,21 @@ impl LocalCellResolver for RuntimeLocalResolver { ) -> Pin>> + Send + 'static>> { let resolver = self.clone(); Box::pin(async move { - let catalog = CellCatalog::new(resolver.layout.clone(), target.tenant()) - .lookup(target.cell_id()) - .await? - .ok_or(Error::Control("target Cell is not cataloged"))?; - let control = CellAuthority::new(resolver.layout) - .load(target.cell_id()) - .await? - .ok_or(Error::Control("target Cell has no authority record"))?; + if resolver.remote_on_miss + && !resolver.runtime.has_local_owner(target.cell_id()).await? + { + // No local actor can execute this request. The peer route + // resolves remote ownership and its receiver fences stale hints. + return Ok(None); + } + let catalog = CellCatalog::new(resolver.layout.clone(), target.tenant()); + let authority = CellAuthority::new(resolver.layout); + let (catalog, control) = tokio::join!( + catalog.lookup(target.cell_id()), + authority.load(target.cell_id()) + ); + let catalog = catalog?.ok_or(Error::Control("target Cell is not cataloged"))?; + let control = control?.ok_or(Error::Control("target Cell has no authority record"))?; resolver.runtime.local_handle(catalog, &control).await }) } diff --git a/crates/cellule-runtime/tests/protocol/client.rs b/crates/cellule-runtime/tests/protocol/client.rs index b4d4025..40febd8 100644 --- a/crates/cellule-runtime/tests/protocol/client.rs +++ b/crates/cellule-runtime/tests/protocol/client.rs @@ -453,6 +453,10 @@ async fn fixture() -> Fixture { } async fn fixture_with_limits(limits: Limits) -> Fixture { + fixture_with_store(limits, Store::new(Arc::new(InMemory::new()))).await +} + +async fn fixture_with_store(limits: Limits, store: Store) -> Fixture { let registry = registry(); let target = CellTarget::new( TenantId::from_bytes([1; 16]), @@ -463,7 +467,6 @@ async fn fixture_with_limits(limits: Limits) -> Fixture { .unwrap(); let cell = target.cell_id(); let incarnation = IncarnationId::from_bytes([4; 16]); - let store = Store::new(Arc::new(InMemory::new())); let layout = CellStorageLayout::new(store, Path::from("client"), [2; 16]); let replica = CellReplica::new( layout.clone(), diff --git a/crates/cellule-runtime/tests/protocol/client/typed.rs b/crates/cellule-runtime/tests/protocol/client/typed.rs index 9e73fbe..5966484 100644 --- a/crates/cellule-runtime/tests/protocol/client/typed.rs +++ b/crates/cellule-runtime/tests/protocol/client/typed.rs @@ -2,6 +2,7 @@ use super::*; use cellule_runtime::cell::actor::CellHandle; +use cellule_store::test_support::CountingObjectStore; #[tokio::test] async fn local_resolver_refusal_never_dispatches_to_the_underlying_owner() { @@ -576,6 +577,101 @@ async fn runtime_client_forwards_to_the_remote_owner() { caller.shutdown().await.unwrap(); fixture.handle().drain().await.unwrap(); } + +#[tokio::test] +async fn remote_runtime_route_skips_local_metadata_without_skipping_local_owner_validation() { + let counted = Arc::new(CountingObjectStore::new(Arc::new(InMemory::new()))); + let fixture = fixture_with_store(Limits::default(), Store::new(counted.clone())).await; + let caller = CellRuntime::new( + SqlWorkerPool::new(1, 4).unwrap(), + 4 * 1024 * 1024, + SessionId::from_bytes([64; 16]), + ) + .unwrap(); + let signer = Arc::new(PeerSigner::new( + SessionId::from_bytes([65; 16]), + fixture.registry.release_digest(), + ed25519_dalek::SigningKey::from_bytes(&[66; 32]), + )); + let round_trip: Arc = Arc::new(LoopbackRoundTrip { + verifier: Arc::new(PeerVerifier::new( + SessionId::from_bytes([65; 16]), + fixture.registry.release_digest(), + signer.verifying_key(), + )), + dispatcher: Arc::new(PeerDispatcher::new( + Arc::clone(&fixture.registry), + Arc::new(LocalResolver { + target: fixture.target.clone(), + handle: fixture.handle().clone(), + }), + Arc::new(RepositoryAuthorizer), + )), + }); + let description = CellDescription { + cell: fixture.target.cell_id(), + incarnation: fixture.incarnation, + code: fixture.registry.module_code(MODULE).unwrap(), + schema: 1, + }; + let principal = PeerPrincipal { + issuer: "https://identity.example".into(), + subject: "alice".into(), + actions: vec!["repository.issue.create".into()], + }; + let remote = CellClient::runtime_with_peer( + Arc::clone(&fixture.registry), + caller.clone(), + fixture.layout.clone(), + Arc::clone(&signer), + principal.clone(), + Arc::clone(&round_trip), + ); + counted.reset(); + assert_eq!( + remote + .query::(&fixture.target, None, ()) + .await + .unwrap() + .output, + 0 + ); + assert_eq!(counted.counts().body_requests(), 0); + + let remote = remote.with_observed_description(description); + counted.reset(); + assert_eq!( + remote + .query::(&fixture.target, None, ()) + .await + .unwrap() + .output, + 0 + ); + assert_eq!(counted.counts().body_requests(), 0); + + let local = CellClient::runtime_with_peer( + Arc::clone(&fixture.registry), + fixture.runtime.as_ref().unwrap().clone(), + fixture.layout.clone(), + signer, + principal, + round_trip, + ) + .with_observed_description(description); + counted.reset(); + assert_eq!( + local + .query::(&fixture.target, None, ()) + .await + .unwrap() + .output, + 0 + ); + assert_eq!(counted.counts().body_requests(), 3); + caller.shutdown().await.unwrap(); + fixture.handle().drain().await.unwrap(); +} #[tokio::test] async fn typed_client_rejects_conflicting_identity_receipt_and_module_before_execution() { let fixture = fixture().await;