From 9c92c1fb04edf710495db7b4c2af32dadcf2d9da Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 1 Oct 2026 00:32:47 -0700 Subject: [PATCH 1/6] Retain durable Git packs and repack serving caches in background --- README.md | 1 + deploy/local-evaluation.md | 138 +++++ docs/evidence/git-maintenance-20260930.json | 17 + .../kubernetes-source-inventory-20260930.json | 21 + .../kubernetes-tree-packed-20260930.json | 141 +++++ docs/kubernetes-qualification.md | 206 +++++++ docs/performance-plan.md | 5 + scripts/benchmark_large_repository.py | 317 +++++++++++ scripts/local_eval.py | 220 ++++++++ scripts/test_local_eval.py | 63 +++ src/deployment/backup/bodies.rs | 13 +- src/git_cache.rs | 123 +++- src/git_cache/maintenance.rs | 533 ++++++++++++++++++ src/git_cache/tests.rs | 154 +++++ src/git_gateway.rs | 213 +++++-- src/git_gateway/fetch.rs | 108 +++- src/git_gateway/hydration.rs | 51 ++ src/git_gateway/maintenance.rs | 59 ++ src/git_http.rs | 8 +- src/git_objects.rs | 86 ++- src/git_objects/tests.rs | 2 +- src/graph.rs | 10 +- src/graph/preparation.rs | 182 +++--- src/lib.rs | 72 ++- src/object_batch.rs | 58 +- src/object_batch/tests.rs | 35 +- src/object_reads.rs | 11 + src/pack_store.rs | 439 +++++++++++++++ src/schema.sql | 15 +- src/server.rs | 5 + src/server/lifecycle.rs | 1 + src/server/residency.rs | 24 +- tests/multi_server/backup.rs | 13 +- tests/multi_server/large_objects.rs | 3 + tests/repository_cell/graph.rs | 9 +- tests/repository_cell/pages.rs | 1 + 36 files changed, 3207 insertions(+), 150 deletions(-) create mode 100644 deploy/local-evaluation.md create mode 100644 docs/evidence/git-maintenance-20260930.json create mode 100644 docs/evidence/kubernetes-source-inventory-20260930.json create mode 100644 docs/evidence/kubernetes-tree-packed-20260930.json create mode 100644 docs/kubernetes-qualification.md create mode 100644 scripts/benchmark_large_repository.py create mode 100644 scripts/local_eval.py create mode 100644 scripts/test_local_eval.py create mode 100644 src/git_cache/maintenance.rs create mode 100644 src/git_gateway/maintenance.rs create mode 100644 src/pack_store.rs diff --git a/README.md b/README.md index a6fe357..332870a 100644 --- a/README.md +++ b/README.md @@ -15,6 +15,7 @@ This page keeps the project overview, setup path, and first repository workflow. | --- | --- | | Understand the storage model | [How Canopy stores a repository](#how-canopy-stores-a-repository) | | Run a development node | [Run a local node](#run-a-local-node) | +| Start an isolated local server and test a Kubernetes-sized repository | [Local evaluation and real-repository benchmark](deploy/local-evaluation.md) | | Create, clone, and push a repository | [Create and use a repository](#create-and-use-a-repository) | | Use the browser, collaboration, or Git LFS | [Use Canopy](docs/user-guide.md) | | Automate repository operations | [API reference](docs/api-reference.md) | diff --git a/deploy/local-evaluation.md b/deploy/local-evaluation.md new file mode 100644 index 0000000..9b2638c --- /dev/null +++ b/deploy/local-evaluation.md @@ -0,0 +1,138 @@ +# Local server and Kubernetes repository qualification + +Use this profile to evaluate the product locally with real Git clients and an +isolated S3-compatible store. The server and provider bind to loopback. It keeps +its credentials, deployment identities, provider volume and reports across +restarts. It is a development install; the native Canopy process does not inherit +the Linux containment ceilings in [the bounded profile](README.md). + +## Start a server + +Prerequisites: Rust 1.97+, Git 2.50.1 or another qualified version, Python 3, +Docker and the AWS CLI. Select a writable volume with enough free space. The +large-repository fixture needs room for compressed durable packs, a native Git +cache, SQLite metadata and graph indexes, and client clones. Blob bodies +from verified receive packs retain Git compression in object storage. Structural +objects remain in SQLite; pushes without a retained native pack still use the +separate streaming store for oversized blobs. Reserve extra disk for ingestion and overlapping cache generations. + +```sh +export CARGO_TARGET_DIR="$PWD/target" +CARGO_INCREMENTAL=0 cargo build --release --locked --bin canopy +python3 scripts/local_eval.py start \ + --binary "$CARGO_TARGET_DIR/release/canopy" \ + --state-dir "$HOME/canopy-evaluation" \ + --disk-gib 64 +``` + +Open . The configured owner is `canopy`. Its bootstrap +credential is `CANOPY_GIT_TOKEN` in the mode-0600 `secrets.json` in the state +directory. The helper never prints credentials. Use that token to sign into the +browser or authenticate Git. The separate signing key and provider credentials +are in the same private file. Keep the state directory outside version control. + +The helper pins its RustFS image by digest and creates an isolated Docker volume, +bucket and application prefix. RustFS has a 2 GiB memory ceiling, a two-CPU +bandwidth ceiling and bounded logs. The native Canopy process has three active +repository slots and a 64 GiB disk admission budget by default; those settings +are evaluation choices, not measured production limits. Neither the Docker +volume nor the host process has a hard filesystem limit in this profile. + +```sh +python3 scripts/local_eval.py status --state-dir "$HOME/canopy-evaluation" +python3 scripts/local_eval.py stop --state-dir "$HOME/canopy-evaluation" +python3 scripts/local_eval.py start --state-dir "$HOME/canopy-evaluation" +``` + +Stop drains the node before stopping its provider. Start reuses the same secrets +and deployment identity. Node startup discards its managed runtime workspace and +restores published state from object storage. The helper checks the process +identity before signaling it and rejects a changed executable: preview schema +and release changes require a new state directory and storage prefix. Do not +replace a running executable or edit release identities to bypass that check. +The native node runs independently of the invoking terminal; run `start` after +a host reboot. Only the provider has automatic Docker restart configured. + +Inspect `server.log` for Canopy diagnostics. `deployment.json` records the exact +executable hash, provider image, container and volume names. The provider volume +is retained by `stop`; removal is a separate, destructive operation. + +## Measure a real repository + +Use a clean, complete source checkout or mirror. The benchmark copies objects +without hard links and changes refs only in its independent fixture. Remote +tracking branches become regular branches; a divergent local branch is retained +alongside an `upstream/` branch. It imports only branches and tags, not GitHub's +issues, reviews, CI state or other hosting metadata. Record which refs the source +actually includes: one complete master history does not cover every release +branch and tag. + +Use a dedicated node for each concurrent benchmark, with a distinct state +directory and port. The recovery stage stops and restarts that node. Start with +the current tree, then test full history: + +```sh +python3 scripts/benchmark_large_repository.py \ + --state-dir "$HOME/canopy-evaluation" \ + --source /path/to/kubernetes \ + --work-dir /path/to/evidence/snapshot \ + --name kubernetes-snapshot --mode snapshot + +python3 scripts/benchmark_large_repository.py \ + --state-dir "$HOME/canopy-evaluation" \ + --source /path/to/kubernetes \ + --work-dir /path/to/evidence/full \ + --name kubernetes --mode full +``` + +Both the work directory and repository name must be new. Snapshot mode creates +one root commit with the source HEAD tree; it cannot establish history capacity. +Full mode preserves reachable history, branches and tags in the source. The +script rejects shallow sources. It imports through stock Git smart HTTP with an +atomic ref update, verifies protocol-v0/v2 mirror clones and strict full `fsck`, +pushes one additional commit, checks incremental fetch and fast-forward pull, +restarts with a fresh managed local workspace, +then compares restored refs and HEAD tree and runs another `fsck`. + +`report.json` is checkpointed at stage boundaries, including on failure. It +records revision, binary/provider identity, source tree, reachable object and +commit counts, stage durations and sampled resource peaks. Operation logs and +`resources.jsonl` remain beside it. Process-tree RSS is sampled every five +seconds and excludes the provider, client processes and filesystem cache; it is +not cgroup memory or a complete peak-memory bound. New runs also record SQLite +and WAL sizes and the indexed object insertion high-water mark. Client timeout +is configurable with `--timeout`; it does not change native server deadlines. + +Treat a failed or timed-out stage as an incomplete gate. Client cancellation +does not prove that an already admitted durable operation has stopped. Inspect +the server and ref state before stopping it or retrying; partially staged objects +may remain in the provider. A passed run establishes correctness only for the +recorded fixture, host and provider. Add filtered/shallow clones, browsing, pulls, +concurrent users, provider faults and bounded Linux measurements before making a +production capacity claim. + +For production work, retain the release gates in [the roadmap](../ROADMAP.md): +schema upgrades, recovery faults, garbage collection, provider qualification, +metrics and alerts, TLS/secret operations and the release artifact pipeline. + +## Background Git maintenance + +Each resident repository checks its serving cache every 60 seconds. It repacks +when the cache has at least 1,024 loose objects or eight pack files. One maintenance +job runs per process, with one native compression thread and separate admission +from user transfers. It writes and verifies a complete new cache generation, +then publishes it only if the object inventory is still current. Existing clones +and fetches retain their original generation until their streams finish. + +The job skips active hydration and retries after concurrent writes. Failure keeps +the old generation available. Shutdown cancels the supervised job and kills its +native process group before repository drain. Server logs record successful +publication, object count, bytes before and after, and elapsed time. Native Git +transfers have a one-hour worker deadline; client disconnection still cancels +streaming workers. Individual provider operations retain their bounded deadlines. + +Durable pack/index files are immutable and backed up alongside external Git/LFS +bodies. Recovery verifies artifact hashes and restores complete packs directly. +The serving-cache repack does not delete durable artifacts or repoint SQLite blob +locators. Durable object-store compaction, orphan collection, and schema upgrades +remain separate production work. Preview schema changes require a fresh prefix. diff --git a/docs/evidence/git-maintenance-20260930.json b/docs/evidence/git-maintenance-20260930.json new file mode 100644 index 0000000..4d3d07d --- /dev/null +++ b/docs/evidence/git-maintenance-20260930.json @@ -0,0 +1,17 @@ +{ + "repository_id": "be13f59a-3ad6-4ceb-b44b-0ae9f30426dd", + "clone_url": "http://127.0.0.1:18082/canopy/maintenance-proof.git", + "push_seconds": [ + 1.973, + 3.482, + 1.935, + 2.779, + 2.591, + 3.391, + 3.671, + 2.856 + ], + "status": "passed", + "publication_log": "2026-10-01T06:57:33.015236Z INFO canopy_server::git_gateway::maintenance: published background Git repack generation repository=be13f59a3ad64cebb44b0ae9f30426dd objects=1616 previous_bytes=356499 packed_bytes=146715 elapsed_seconds=0.375848458", + "clone_seconds": 0.915 +} diff --git a/docs/evidence/kubernetes-source-inventory-20260930.json b/docs/evidence/kubernetes-source-inventory-20260930.json new file mode 100644 index 0000000..ababa38 --- /dev/null +++ b/docs/evidence/kubernetes-source-inventory-20260930.json @@ -0,0 +1,21 @@ +{ + "counts": { + "commit": 162267, + "tree": 1066880, + "blob": 582504, + "tag": 1247 + }, + "expanded_bytes": { + "commit": 100149074, + "tree": 870535081, + "blob": 34105250811, + "tag": 233534 + }, + "above_inline_limit_counts": { + "blob": 6124 + }, + "above_inline_limit_bytes": { + "blob": 18588908612 + }, + "seconds": 17.579 +} diff --git a/docs/evidence/kubernetes-tree-packed-20260930.json b/docs/evidence/kubernetes-tree-packed-20260930.json new file mode 100644 index 0000000..4ec8a66 --- /dev/null +++ b/docs/evidence/kubernetes-tree-packed-20260930.json @@ -0,0 +1,141 @@ +{ + "mode": "snapshot", + "name": "kubernetes-tree-packed", + "status": "passed", + "stages": [ + { + "name": "copy-source", + "status": "passed", + "exit_code": 0, + "seconds": 70.012 + }, + { + "name": "reachable-objects", + "status": "passed", + "exit_code": 0, + "seconds": 5.894 + }, + { + "name": "push", + "status": "passed", + "exit_code": 0, + "seconds": 127.567 + }, + { + "name": "warm-clone-v0", + "status": "passed", + "exit_code": 0, + "seconds": 9.799 + }, + { + "name": "warm-fsck-v0", + "status": "passed", + "exit_code": 0, + "seconds": 1.512 + }, + { + "name": "warm-clone-v2", + "status": "passed", + "exit_code": 0, + "seconds": 8.292 + }, + { + "name": "warm-fsck-v2", + "status": "passed", + "exit_code": 0, + "seconds": 1.71 + }, + { + "name": "prepare-pull-client", + "status": "passed", + "exit_code": 0, + "seconds": 34.278 + }, + { + "name": "configure-pull-client", + "status": "passed", + "exit_code": 0, + "seconds": 0.04 + }, + { + "name": "incremental-prepare", + "status": "passed", + "exit_code": 0, + "seconds": 0.042 + }, + { + "name": "advance-default-branch", + "status": "passed", + "exit_code": 0, + "seconds": 0.132 + }, + { + "name": "incremental-push", + "status": "passed", + "exit_code": 0, + "seconds": 1.272 + }, + { + "name": "incremental-fetch", + "status": "passed", + "exit_code": 0, + "seconds": 0.709 + }, + { + "name": "incremental-pull", + "status": "passed", + "exit_code": 0, + "seconds": 0.749 + }, + { + "name": "incremental-fsck", + "status": "passed", + "exit_code": 0, + "seconds": 1.728 + }, + { + "name": "fresh-workspace-restart", + "status": "passed", + "seconds": 7.269 + }, + { + "name": "cold-clone", + "status": "passed", + "exit_code": 0, + "seconds": 7.293 + }, + { + "name": "cold-fsck", + "status": "passed", + "exit_code": 0, + "seconds": 2.152 + } + ], + "canopy_binary_sha256": "e67bd56807df3fde4b8cd2dc1f034216ee38ff920fa8dd3ee9ad908c75779d8f", + "canopy_revision": "9c5f1d1bf837fdc2e38ee229aa409712b9579f69", + "canopy_source_tree_sha256": "69231818c4e082fb35e5655fd3274b2058aef720dfd9580de95569395dd34945", + "canopy_working_tree_dirty": true, + "provider_image": "ghcr.io/rustfs/rustfs@sha256:0c3c7030ffb93afde8d359fb1db957b85033ede05115518bd0dede51f4353f6a", + "git_version": "git version 2.50.1 (Apple Git-155)", + "host_platform": "macOS-26.5.2-arm64-arm-64bit-Mach-O", + "host_cpu_count": 12, + "disk_limit_bytes": 51539607552, + "active_repository_limit": 3, + "node_process_tree_peak_rss_bytes": 125992960, + "started_at_utc": "2026-10-01T06:48:17Z", + "source_head": "6d805ebe018f428d503fdc01bb6d57bd9574f598", + "source_tree": "bea8806e82f4af68e950de7c32665381a7add5d7", + "default_branch": "refs/heads/master", + "ref_count": 1, + "reachable_commits": 1, + "reachable_objects": 29839, + "repository_id": "f8accc8a-d405-4a8c-95c5-ec1d74224a73", + "clone_url": "http://127.0.0.1:18082/canopy/kubernetes-tree-packed.git", + "peak_repository_sqlite_and_wal_bytes": 16854496, + "evaluation_branch": "refs/heads/canopy-evaluation-f8accc8a-d405-4a8c-95c5-ec1d74224a73", + "incremental_commit": "8ef8cb7d7a7c3eb5d6db26a57c3572e4cd4f70a5", + "restored_ref_count": 2, + "restored_tree": "bea8806e82f4af68e950de7c32665381a7add5d7", + "finished_at_utc": "2026-10-01T06:53:02Z", + "source_repository": "https://github.com/kubernetes/kubernetes" +} diff --git a/docs/kubernetes-qualification.md b/docs/kubernetes-qualification.md new file mode 100644 index 0000000..c668bc4 --- /dev/null +++ b/docs/kubernetes-qualification.md @@ -0,0 +1,206 @@ +# Kubernetes qualification progress, 2026-09-30–2026-10-01 + +The new local evaluation server is running at . +The Kubernetes **tree** gate now passes, including recovery. The complete +upstream history gate is running; it is not yet a capacity qualification. +Earlier failures are retained below for comparison. + +The measured predecessor release binary (small packed blobs) is +`e67bd56807df3fde4b8cd2dc1f034216ee38ff920fa8dd3ee9ad908c75779d8f`, +based on revision `9c5f1d1bf837fdc2e38ee229aa409712b9579f69`. +Its source fingerprint and provider identity are in each report. The complete +source mirror includes 1,312 branches/tags and 162,267 reachable commits, +with 1,812,898 reachable objects at HEAD +`6d805ebe018f428d503fdc01bb6d57bd9574f598`. + +The implementation retains verified receive packs and indexes durably, with canonical +identities and graph certificates in SQLite. The measured tree build keeps small blobs compressed. The current PR extends +this to oversized packed blobs with bounded streaming verification, reads packs +in physical order, and increases metadata/certificate batching; these newer +changes still require the complete-history gate. Recovery restores +verified packs and uses a durable coverage proof rather than reconstructing +all previously covered objects. Existing object identities and digests are +checked before approving a pack, including thin-pack bases. + +| Current fixture | Result | +| --- | --- | +| Pristine Kubernetes HEAD tree: 29,839 objects, one synthetic root | Passed: push 127.567 s; warm clones v0 9.799 s, v2 8.292 s; incremental push 1.272 s, fetch 0.709 s, pull 0.749 s; cold clone 7.293 s; all strict full `fsck` and ref/tree comparisons passed | +| Eight-push compressed history: 1,616 objects | Background repack published in 0.376 s, reducing serving-cache bytes 356,499 → 146,715; subsequent clone 0.915 s and strict `fsck` passed | +| Complete upstream Kubernetes history | Running; no passing claim yet | + +The tree fixture's SQLite file is about 10.13 MiB after recovery, and sampled +node-plus-descendant RSS peaked at 120.16 MiB. Timings include competition from +other builds and repository evaluations on this host; they are not an isolated +throughput or concurrency SLA measurement. + +Verification of the measured predecessor: 113 library tests passed with one test thread; the repository Cell +integration gate passed; filtered-clone tests passed (two tests); the compressed +pack backup/restore gate passed after deleting original storage. The initial +concurrent library run had one native-fence cleanup assertion fail; the serial +run passed. This does not establish cancellation behavior under arbitrary +concurrent process creation. Four local-evaluation helper tests passed. + +Maintenance checks each resident repository every 60 seconds, triggers at 1,024 +loose objects or eight pack files, and admits one job per process. It uses one +native compression thread, verifies a new generation, preserves active reader +ownership, and skips publication if concurrent writes change its inventory. +It has supervised shutdown cancellation and separate admission from transfers. +Native Git transfers now have a one-hour worker deadline. + +Durable artifacts and indexes are included in backup verification/copy/restore. +Serving-cache repack does not delete durable packs or repoint blob locators. +Durable pack compaction/garbage collection, preview schema migration, provider +fault tests and bounded Linux concurrent-load qualification remain production +work. See [the operational recipe](../deploy/local-evaluation.md). + +Current evidence: + +```text +/Volumes/Workspace/crabbuild-target/canopy-scale-eval-20260930/ + packed-kubernetes-snapshot/report.json + packed-kubernetes-full/report.json + packed-kubernetes-full/metadata-latency.jsonl + maintenance-proof/report.json + packed-service/server.log + packed-lib-serial.log + packed-backup-test.log + packed-partial-test.log +``` + +## Earlier evaluation of the original release + +**Result: this build did not pass Kubernetes-sized repository hosting on the +local evaluation host.** The clean full-history push hit the native HTTP Git +worker deadline before durable object ingestion. A separate Kubernetes tree +fixture imported successfully, but its full clone failed. Neither large fixture +completed the recovery gate. These are observed failures for this environment, +not a universal maximum repository size. + +## Build and environment + +- Canopy revision: `9c5f1d1bf837fdc2e38ee229aa409712b9579f69`. +- Cellule pin: `a3fbfb0115a1ae2519ee8f8e0cf6b8e72fdaa303`. +- Release binary SHA-256: `e44148670da22af1a870837cdbec48fe633f2c09561b3870c20c65d828958ed0`. +- Rust/Cargo 1.97.0; Apple Git 2.50.1; 12-CPU, 32-GiB Apple Silicon macOS host. +- Workspace: external USB APFS SSD. Other repository evaluations were running + on the host; this was not an isolated throughput benchmark. +- Provider: local ARM64 Docker RustFS image + `ghcr.io/rustfs/rustfs@sha256:0c3c7030ffb93afde8d359fb1db957b85033ede05115518bd0dede51f4353f6a`, + with a dedicated volume and application prefix per node. Provider limits: + 2 GiB memory, two CPUs, 256 processes, bounded logs. The Docker VM has eight + CPUs and about 16 GiB memory. +- Each native Canopy node: three active repository slots, 64 GiB disk admission + budget. No hard cgroup CPU/memory/filesystem containment for the host process. +- Large tests ran on separate nodes. The full-history node was stopped after + recording its failure. The evaluation node remains at + , with credentials in its private state directory. + +## Measured results + +| Fixture | Scope | Result | +| --- | --- | --- | +| Clean Kubernetes history | Source HEAD `08147af84478f859c2e2234d71ceace8bdb412c7`; 141,666 reachable commits, 1,663,509 reachable objects; source packed object store about 1.22 GiB | Atomic push rejected in 265.747 s; server recorded `Http(Timeout)` | +| Kubernetes tree fixture | Source HEAD `1124a801ebcedde8880b3cb9a4721745bad55c4c`; one synthetic root commit, 30,021 reachable objects | Push passed in 314.899 s; protocol-v0 mirror clone failed in 298.077 s with truncated pack/RPC failure | +| Small welcome/recovery fixture | Three initial objects, one initial commit, then one additional commit | Protocol-v0/v2 clones, strict full `fsck`, incremental push, fresh-workspace restart, cold clone and restored ref/tree comparison all passed | + +The clean source contains complete master history, not the complete set of +GitHub release branches, tags or hosting metadata. Its copied fixture also +contained a redundant remote HEAD alias as a branch; both fixture refs were +rejected. The benchmark helper now drops that alias during normalization. +The tree fixture comes from an older local Kubernetes checkout with local +changes (`Superset-arm64.dmg` and `crab.toml` in its tip commit). It is a scoped +tree fixture, not pristine upstream history or a passing large-repository gate. + +The full-history server logged: + +```text +Git push failed before publication ... error=Http(Timeout) +``` + +The associated native cache cleanup reported `operation would block` and retained +disk admission until process restart. The client received an unpack rejection; +the object insertion high-water mark remained unset during the sampled run. +The native worker deadline is **120 seconds**, beginning after request spooling +and cache preparation; the 265.747-second client duration includes those other +phases. The tree clone's precise terminal server error was not logged, so its +cause remains unconfirmed. Do not label that clone failure a proven timeout. + +Sampled node-plus-descendant RSS peaked at 222.5 MiB for the full-history attempt +and 153.25 MiB for the tree fixture. These samples exclude the Git client, +provider and filesystem cache, and do not capture every transient peak. The +tree repository database reached about 239 MiB; its managed local workspace +reached about 412 MiB during clone preparation. Those figures do not size the +full-history database because that import never reached durable ingestion. + +While the tree clone and separate full import ran, ten sequential requests per +endpoint all returned 200: + +| Endpoint | Median | Maximum | +| --- | --- | --- | +| Readiness | 0.47 ms | 28.55 ms | +| Repository metadata | 1.32 ms | 354.20 ms | +| Root tree browser | 1.92 ms | 143.80 ms | + +Ten samples establish basic responsiveness in that phase, not p95/p99 service +levels or concurrent-user capacity. + +## Evidence and reproduction + +The raw reports and logs are retained under: + +```text +/Volumes/Workspace/crabbuild-target/canopy-product-eval-9c5f1d1-20260930/ + full/report.json + full/push.log + full/resources.jsonl + full-service/server.log + snapshot-v2/report.json + snapshot-v2/warm-clone-v0.log + snapshot-v2/resources.jsonl + metadata-probe.json + welcome-proof-v2/report.json + main-proof/report.json +``` + +Use [the local evaluation recipe](../deploy/local-evaluation.md) and +`scripts/benchmark_large_repository.py` with a fresh repository name and work +directory. The helper rejects shallow source repositories and records exact +refs and source trees. Reports distinguish a failed stage from a completed +clone/recovery gate; do not advertise a passing push as end-to-end support. + +The passing small recovery run also exercises remote HEAD normalization. The +setup helper's restart probe initially rejected the port while accepted +connections were in `TIME_WAIT`; it now uses address reuse, and the complete +recovery run passed after that fix. Four focused helper tests passed for identity +reuse, private credentials/environment isolation, unrelated-directory rejection, +stale PID protection and changed-executable rejection. +The same complete small-repository gate also passed with a bare source on +`main` and no remote-tracking refs, exercising the helper's default-branch and +empty-normalization paths. + +## Product work this result prioritizes + +1. Make native worker timeout policy explicit and suitable for large pack + indexing/generation. Preserve cancellation, descendant cleanup and resource + admission. Merely increasing the deadline does not establish capacity. +2. Record terminal streaming failures and timings for request spooling, native + Git, object ingestion, graph certification, cache hydration and pack output. + The failed clone currently lacks a terminal diagnostic in the server log. +3. Measure and optimize object ingestion and cold/warm cache rebuilding. + Ordinary objects are expanded into SQLite and reconstructed into a native + cache; compressed pack size is a poor deployment sizing estimate. The + preliminary all-ref local corpus had about 32 GiB of logical object bodies + despite about 1.29 GiB packed storage; this is a different corpus from the + clean master-history test. +4. Re-run pristine master history, release branches/tags, protocol-v0/v2, + filtered/shallow clones, incremental pushes and fresh-workspace recovery. + Then qualify concurrent traffic and the intended bounded Linux/provider + configuration. Keep those claims separate. +5. Qualify collaboration traversal independently. Merge-base and ancestry + searches currently stop at 100,000 discovered commits or 250,000 edges; + some operations on this history can exceed those limits. No failing + pull-request traversal was demonstrated in this evaluation. + +Continue the remaining production gates in [the roadmap](../ROADMAP.md), +especially migrations, garbage collection, provider/failure qualification, +operational metrics and alerts, TLS/credential operations and release artifacts. diff --git a/docs/performance-plan.md b/docs/performance-plan.md index 2247296..6655684 100644 --- a/docs/performance-plan.md +++ b/docs/performance-plan.md @@ -15,6 +15,11 @@ flowchart LR ## Read the result before the target +The [Kubernetes evaluation](kubernetes-qualification.md) records the original +large-import failures and the new passing tree/recovery and maintenance gates. +Complete upstream history qualification is still running. A passing tree fixture +does not establish full-history capacity. + | Question | Current answer | Where to verify it | | --- | --- | --- | | Can large Git and LFS bodies survive fresh-disk recovery? | Yes, in a release-mode local RustFS run on an earlier Cellule pin with a Git blob and LFS object above 5 GiB | [Recorded large-transfer qualification](#recorded-large-transfer-qualification) | diff --git a/scripts/benchmark_large_repository.py b/scripts/benchmark_large_repository.py new file mode 100644 index 0000000..90e1b46 --- /dev/null +++ b/scripts/benchmark_large_repository.py @@ -0,0 +1,317 @@ +#!/usr/bin/env python3 +"""Qualify a real repository against local_eval.py, with a durable JSON report. + +Imports branches/tags, checks warm stock Git clones, then restarts the node +(which discards its runtime workspace) and verifies a cold clone and fsck. +Snapshot mode imports one root commit with the source HEAD tree, not history. +Run on a dedicated evaluation node: the recovery stage restarts that node. +""" +import argparse +import base64 +import hashlib +import json +import os +import platform +from pathlib import Path +import signal +import sqlite3 +import subprocess +import sys +import threading +import time +import urllib.request + +import local_eval + + +def references(directory): + output = local_eval.run("git", "-C", str(directory), "for-each-ref", + "--format=%(objectname) %(refname)", "refs/heads", "refs/tags") + return dict(line.split(" ", 1)[::-1] for line in output.splitlines()) + + +def update_refs(directory, commands): + if not commands: + return + subprocess.run(["git", "-C", str(directory), "update-ref", "--stdin"], + input="option no-deref\n" + "\n".join(commands) + "\n", + text=True, check=True, capture_output=True) + + +def process_memory(pid): + output = local_eval.run("ps", "-axo", "pid=,ppid=,rss=") + processes = [tuple(map(int, line.split())) for line in output.splitlines()] + family = {pid} + while True: + children = {child for child, parent, _ in processes if parent in family} + added = children - family + if not added: + break + family.update(added) + return sum(rss * 1024 for child, _, rss in processes if child in family) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--state-dir", required=True, type=Path) + parser.add_argument("--source", required=True, type=Path) + parser.add_argument("--work-dir", required=True, type=Path) + parser.add_argument("--name", default="kubernetes") + parser.add_argument("--mode", choices=("full", "snapshot"), default="full") + parser.add_argument("--timeout", type=int, default=3600, help="seconds per Git operation") + args = parser.parse_args() + args.state_dir = args.state_dir.resolve() + args.work_dir = args.work_dir.resolve() + args.work_dir.mkdir(parents=True, exist_ok=False) + metadata = json.loads((args.state_dir / "deployment.json").read_text()) + config = json.loads((args.state_dir / "config.json").read_text()) + env = local_eval.environment(args.state_dir) + auth = base64.b64encode((config["owner"] + ":" + env["CANOPY_GIT_TOKEN"]).encode()).decode() + git_env = {k: v for k, v in os.environ.items() if not k.startswith("GIT_")} + git_env.update(GIT_CONFIG_COUNT="2", GIT_CONFIG_KEY_0="credential.helper", + GIT_CONFIG_VALUE_0="", GIT_CONFIG_KEY_1="http.extraHeader", + GIT_CONFIG_VALUE_1="Authorization: Basic " + auth, + GIT_TERMINAL_PROMPT="0", GIT_LFS_SKIP_SMUDGE="1") + source_digest = hashlib.sha256() + for source_file in sorted([*Path("src").rglob("*.rs"), *Path("src").rglob("*.sql"), Path("Cargo.toml"), Path("Cargo.lock")]): + source_digest.update(str(source_file).encode() + b"\0" + source_file.read_bytes()) + report = { + "mode": args.mode, "name": args.name, "status": "running", "stages": [], + "source": str(args.source.resolve()), "canopy_binary_sha256": metadata["binary_sha256"], + "canopy_revision": local_eval.run("git", "rev-parse", "HEAD"), + "canopy_source_tree_sha256": source_digest.hexdigest(), + "canopy_working_tree_dirty": bool(local_eval.run("git", "status", "--porcelain")), + "provider_image": metadata["provider_image"], "git_version": local_eval.run("git", "--version"), + "host_platform": platform.platform(), "host_cpu_count": os.cpu_count(), + "disk_limit_bytes": config["local_disk_limit_bytes"], + "active_repository_limit": config["max_active_repositories"], + "node_process_tree_peak_rss_bytes": 0, + "started_at_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), + } + report_path = args.work_dir / "report.json" + stopped = threading.Event() + + def sample(): + with (args.work_dir / "resources.jsonl").open("w") as output: + while not stopped.is_set(): + pid = local_eval.owned_pid(args.state_dir, metadata) + if pid: + try: + rss = process_memory(pid) + report["node_process_tree_peak_rss_bytes"] = max( + report["node_process_tree_peak_rss_bytes"], rss) + output.write(json.dumps({"time": time.time(), "pid": pid, + "process_tree_rss_bytes": rss}) + "\n") + # MAX(sequence) uses the integer primary key. Avoid + # table scans or long transactions on the live Cell. + for database in (args.state_dir / "node" / "runtime-v1").glob("*/repository.sqlite"): + try: + connection = sqlite3.connect(database.as_uri() + "?mode=ro", uri=True, timeout=0.2) + try: + high_water = connection.execute("SELECT max(sequence) FROM objects").fetchone()[0] + finally: + connection.close() + size = database.stat().st_size + wal = Path(str(database) + "-wal") + wal_size = wal.stat().st_size if wal.exists() else 0 + output.write(json.dumps({"time": time.time(), + "repository": database.parent.name, "object_sequence": high_water, + "sqlite_bytes": size, "sqlite_wal_bytes": wal_size}) + "\n") + report["peak_repository_sqlite_and_wal_bytes"] = max( + report.get("peak_repository_sqlite_and_wal_bytes", 0), size + wal_size) + except (OSError, sqlite3.Error): + pass + output.flush() + except (OSError, subprocess.CalledProcessError): + pass + stopped.wait(5) + + def checkpoint(): + local_eval.save(report_path, report) + + def git(stage, *arguments, cwd=None): + entry = {"name": stage, "status": "running"} + report["stages"].append(entry) + checkpoint() + print(f"{stage}: started", flush=True) + started = time.monotonic() + try: + with (args.work_dir / (stage + ".log")).open("wb") as output: + child = subprocess.Popen(["git", *arguments], cwd=cwd, env=git_env, + stdout=output, stderr=output, start_new_session=True) + try: + code = child.wait(timeout=args.timeout) + except BaseException: + # Stop the whole client worker group, including pack-objects. + os.killpg(child.pid, signal.SIGTERM) + try: + child.wait(timeout=10) + except subprocess.TimeoutExpired: + os.killpg(child.pid, signal.SIGKILL) + child.wait() + raise + entry["exit_code"] = code + if code: + raise RuntimeError(f"{stage} failed; see {stage}.log") + entry["status"] = "passed" + except BaseException: + entry["status"] = "failed" + raise + finally: + entry["seconds"] = round(time.monotonic() - started, 3) + checkpoint() + print(f"{stage}: {entry['status']} ({entry['seconds']}s)", flush=True) + + def api(path, payload=None, method=None): + request = urllib.request.Request(metadata["url"] + path, + data=None if payload is None else json.dumps(payload).encode(), method=method, + headers={"Authorization": "Bearer " + env["CANOPY_GIT_TOKEN"], + "Content-Type": "application/json"}) + with urllib.request.urlopen(request, timeout=60) as response: + return json.load(response) + + sampler = threading.Thread(target=sample, daemon=True) + sampler.start() + checkpoint() + try: + if not local_eval.ready(metadata["url"]): + raise RuntimeError("evaluation node must be ready before benchmarking") + if local_eval.run("git", "-C", str(args.source), "rev-parse", "--is-shallow-repository") != "false": + raise RuntimeError("source must contain full history; shallow source is not a scale gate") + report["source_head"] = local_eval.run("git", "-C", str(args.source), "rev-parse", "HEAD") + report["source_tree"] = local_eval.run("git", "-C", str(args.source), "rev-parse", "HEAD^{tree}") + mirror = args.work_dir / "source.git" + git("copy-source", "clone", "--mirror", "--no-hardlinks", str(args.source.resolve()), str(mirror)) + # A normal local checkout has remote-tracking release branches. Preserve + # those as ordinary branches in the independent import fixture. + remote = local_eval.run("git", "-C", str(mirror), "for-each-ref", + "--format=%(objectname) %(refname) %(symref)", "refs/remotes/origin") + local_branches = references(mirror) + changes = [] + for line in remote.splitlines(): + parts = line.split() + oid, reference = parts[:2] + if len(parts) == 2 and reference != "refs/remotes/origin/HEAD": + branch = reference.replace("refs/remotes/origin/", "refs/heads/", 1) + existing = local_branches.get(branch) + if existing and existing != oid: + # A checkout can legitimately lag origin. Retain both tips, + # keeping its HEAD tree as the independently verified tree. + branch = reference.replace("refs/remotes/origin/", "refs/heads/upstream/", 1) + if branch in local_branches: + raise RuntimeError("upstream branch naming collision; use a source mirror") + changes.append(f"update {branch} {oid}") + changes.append(f"delete {reference}") + update_refs(mirror, changes) + if args.mode == "snapshot": + commit_env = {**git_env, "GIT_AUTHOR_NAME": "Canopy Evaluation", + "GIT_AUTHOR_EMAIL": "evaluation@example.invalid", + "GIT_COMMITTER_NAME": "Canopy Evaluation", + "GIT_COMMITTER_EMAIL": "evaluation@example.invalid"} + commit = local_eval.run("git", "-C", str(mirror), "commit-tree", + report["source_tree"], "-m", "Kubernetes HEAD snapshot (no history)", env=commit_env) + # Deletion and update of master must be separate update-ref batches. + update_refs(mirror, [f"delete {ref}" for ref in references(mirror)]) + update_refs(mirror, [f"update refs/heads/master {commit}"]) + local_eval.run("git", "-C", str(mirror), "symbolic-ref", "HEAD", "refs/heads/master") + expected = references(mirror) + default_ref = local_eval.run("git", "-C", str(mirror), "symbolic-ref", "HEAD") + if default_ref not in expected or expected[default_ref] != ( + commit if args.mode == "snapshot" else report["source_head"]): + raise RuntimeError("source HEAD must name an imported branch") + report["default_branch"] = default_ref + local_eval.save(args.work_dir / "expected-refs.json", expected) + report["ref_count"] = len(expected) + report["reachable_commits"] = int(local_eval.run("git", "-C", str(mirror), "rev-list", "--count", "--branches", "--tags")) + git("reachable-objects", "-C", str(mirror), "rev-list", "--objects", "--branches", "--tags") + with (args.work_dir / "reachable-objects.log").open("rb") as objects: + report["reachable_objects"] = sum(1 for _ in objects) + checkpoint() + created = api("/api/repositories", {"name": args.name}) + report["repository_id"] = created["repository_id"] + url = created["clone_url"] + report["clone_url"] = url + git("push", "-C", str(mirror), "push", "--atomic", url, + "refs/heads/*:refs/heads/*", "refs/tags/*:refs/tags/*") + endpoint = f"/api/repositories/{args.name}/default-branch" + head = api(endpoint) + api(endpoint, {"repository_id": created["repository_id"], "reference": default_ref, + "expected_generation": head["generation"]}, "PUT") + for protocol in ("0", "2"): + clone = args.work_dir / ("warm-v" + protocol + ".git") + git("warm-clone-v" + protocol, "-c", "protocol.version=" + protocol, + "clone", "--mirror", url, str(clone)) + if references(clone) != expected: + raise RuntimeError("warm clone refs differ from the import fixture") + git("warm-fsck-v" + protocol, "-C", str(clone), "fsck", "--strict", "--full") + pull_work = args.work_dir / "pull-work" + git("prepare-pull-client", "clone", "--shared", "--single-branch", "--branch", + default_ref.removeprefix("refs/heads/"), str(args.work_dir / "warm-v2.git"), str(pull_work)) + git("configure-pull-client", "-C", str(pull_work), "remote", "set-url", "origin", url) + # The binary's startup contract removes runtime-v1 before recovering + # authoritative Cells from the provider; no manual deletion is needed. + commit_env = {**git_env, "GIT_AUTHOR_NAME": "Canopy Evaluation", + "GIT_AUTHOR_EMAIL": "evaluation@example.invalid", + "GIT_COMMITTER_NAME": "Canopy Evaluation", + "GIT_COMMITTER_EMAIL": "evaluation@example.invalid"} + increment = local_eval.run("git", "-C", str(mirror), "commit-tree", + report["source_tree"], "-p", default_ref, "-m", "Incremental evaluation commit", + env=commit_env) + incremental_ref = "refs/heads/canopy-evaluation-" + created["repository_id"] + if incremental_ref in expected: + raise RuntimeError("evaluation branch collides with imported refs") + report["evaluation_branch"] = incremental_ref + git("incremental-prepare", "-C", str(mirror), "update-ref", incremental_ref, increment) + git("advance-default-branch", "-C", str(mirror), "update-ref", default_ref, increment) + git("incremental-push", "-C", str(mirror), "push", "--atomic", url, + incremental_ref + ":" + incremental_ref, default_ref + ":" + default_ref) + git("incremental-fetch", "-C", str(args.work_dir / "warm-v2.git"), "fetch", "origin", "--prune") + git("incremental-pull", "-C", str(pull_work), "pull", "--ff-only", "origin", default_ref.removeprefix("refs/heads/")) + if local_eval.run("git", "-C", str(pull_work), "rev-parse", "HEAD") != increment: + raise RuntimeError("pull client did not reach the incremental commit") + git("incremental-fsck", "-C", str(pull_work), "fsck", "--strict", "--full") + expected = references(mirror) + report["incremental_commit"] = increment + local_eval.save(args.work_dir / "expected-restored-refs.json", expected) + checkpoint() + restart = {"name": "fresh-workspace-restart", "status": "running"} + report["stages"].append(restart) + checkpoint() + started = time.monotonic() + try: + subprocess.run([sys.executable, str(Path(local_eval.__file__)), "stop", + "--state-dir", str(args.state_dir)], check=True) + subprocess.run([sys.executable, str(Path(local_eval.__file__)), "start", + "--state-dir", str(args.state_dir)], check=True) + restart["status"] = "passed" + except BaseException: + restart["status"] = "failed" + raise + finally: + restart["seconds"] = round(time.monotonic() - started, 3) + checkpoint() + cold = args.work_dir / "cold.git" + git("cold-clone", "clone", "--mirror", url, str(cold)) + if references(cold) != expected: + raise RuntimeError("cold clone refs differ after object-store recovery") + git("cold-fsck", "-C", str(cold), "fsck", "--strict", "--full") + tree = local_eval.run("git", "-C", str(cold), "rev-parse", default_ref + "^{tree}") + if tree != report["source_tree"]: + raise RuntimeError("restored Kubernetes HEAD tree differs from source") + report["restored_ref_count"] = len(expected) + report["restored_tree"] = tree + report["status"] = "passed" + except BaseException as error: + report["status"] = "failed" + report["error"] = type(error).__name__ + ": " + str(error) + raise + finally: + stopped.set() + sampler.join(timeout=10) + report["finished_at_utc"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()) + checkpoint() + print(str(report_path), flush=True) + + +if __name__ == "__main__": + main() diff --git a/scripts/local_eval.py b/scripts/local_eval.py new file mode 100644 index 0000000..3fcab71 --- /dev/null +++ b/scripts/local_eval.py @@ -0,0 +1,220 @@ +#!/usr/bin/env python3 +"""Start/stop an isolated, persistent local Canopy evaluation node and RustFS. + +Credentials stay in a mode-0600 file outside the checkout. This development +profile binds only loopback; it does not establish Linux container containment. +""" +import argparse +import hashlib +import json +import os +from pathlib import Path +import secrets +import signal +import socket +import subprocess +import time +import urllib.request +import uuid + +PROVIDER_IMAGE = "ghcr.io/rustfs/rustfs@sha256:0c3c7030ffb93afde8d359fb1db957b85033ede05115518bd0dede51f4353f6a" + + +def run(*args, env=None): + return subprocess.run(args, env=env, check=True, capture_output=True, + text=True).stdout.strip() + + +def save(path, value, private=False): + temporary = path.with_suffix(path.suffix + ".tmp") + fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600) + os.fchmod(fd, 0o600) + with os.fdopen(fd, "w") as output: + json.dump(value, output, indent=2) + output.write("\n") + temporary.replace(path) + if not private: + path.chmod(0o644) + + +def environment(state): + # Never inherit unrelated provider credentials or endpoints. + env = {k: v for k, v in os.environ.items() if not k.startswith("AWS_")} + env.update(json.loads((state / "secrets.json").read_text())) + env.update(AWS_DEFAULT_REGION="us-east-1", AWS_REGION="us-east-1", + AWS_EC2_METADATA_DISABLED="true", AWS_CONFIG_FILE=os.devnull, + AWS_SHARED_CREDENTIALS_FILE=os.devnull, AWS_ALLOW_HTTP="true", + RUST_LOG="canopy=info,canopy_server=info,cellule_runtime=warn,cellule_ltx=warn") + return env + + +def owned_pid(state, metadata): + try: + pid = int((state / "server.pid").read_text()) + command = run("ps", "-p", str(pid), "-o", "command=") + if str(state / "config.json") in command and metadata["binary"] in command: + return pid + except (OSError, ValueError, subprocess.CalledProcessError): + pass + return None + + +def ready(url): + try: + with urllib.request.urlopen(url + "/readyz", timeout=2) as response: + return response.status == 200 + except OSError: + return False + + +def initialize(args): + state = args.state_dir + if (state / "deployment.json").exists(): + return json.loads((state / "deployment.json").read_text()) + if any(state.iterdir()): + raise RuntimeError("state directory is not empty; use a new directory") + if not args.binary: + raise RuntimeError("first start requires --binary pointing to a release build") + binary = args.binary.resolve() + if not binary.is_file(): + raise RuntimeError("build canopy before starting the evaluation node") + identifier = uuid.uuid4().hex[:12] + metadata = { + "binary": str(binary), "binary_sha256": hashlib.sha256(binary.read_bytes()).hexdigest(), + "container": "canopy-eval-" + identifier, + "volume": "canopy-eval-" + identifier + "-data", + "provider_image": args.provider_image, + "bucket": "canopy-eval", "url": f"http://127.0.0.1:{args.port}", + } + credentials = { + "CANOPY_GIT_TOKEN": "cnp_" + secrets.token_hex(32), + "CANOPY_NODE_SIGNING_KEY_HEX": secrets.token_hex(32), + "AWS_ACCESS_KEY_ID": "canopy-" + identifier, + "AWS_SECRET_ACCESS_KEY": secrets.token_hex(32), + } + config = { + "storage_url": "s3://canopy-eval/" + identifier, + "tenant_id": str(uuid.uuid4()), "application_id": str(uuid.uuid4()), + "node_id": str(uuid.uuid4()), "fleet_digest": secrets.token_hex(32), + "image_digest": metadata["binary_sha256"], "owner": "canopy", + "public_url": metadata["url"], "listen": f"127.0.0.1:{args.port}", + "peer_endpoint": "https://local-evaluation.example.invalid", + "data_dir": str(state / "node"), + "local_disk_limit_bytes": args.disk_gib * 1024**3, + "max_active_repositories": args.active_repositories, + } + save(state / "secrets.json", credentials, private=True) + save(state / "config.json", config) + save(state / "deployment.json", metadata) + return metadata + + +def start(args, metadata): + state = args.state_dir + if owned_pid(state, metadata): + if not ready(metadata["url"]): + raise RuntimeError("node process exists but is not ready; inspect server.log") + return + binary = Path(metadata["binary"]) + if hashlib.sha256(binary.read_bytes()).hexdigest() != metadata["binary_sha256"]: + raise RuntimeError("binary changed; this preview requires a fresh storage prefix/state directory") + with socket.socket() as probe: + # A drained listener can leave accepted connections in TIME_WAIT. + # Match the server's normal address reuse without permitting a second + # live listener on the same endpoint. + probe.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + probe.bind(("127.0.0.1", int(metadata["url"].rsplit(":", 1)[1]))) + credentials = json.loads((state / "secrets.json").read_text()) + exists = subprocess.run(["docker", "container", "inspect", metadata["container"]], + capture_output=True).returncode == 0 + if exists: + run("docker", "start", metadata["container"]) + else: + run("docker", "volume", "create", metadata["volume"]) + # Pass values through docker's inherited environment, not its argv. + provider_env = {**os.environ, "RUSTFS_ACCESS_KEY": credentials["AWS_ACCESS_KEY_ID"], + "RUSTFS_SECRET_KEY": credentials["AWS_SECRET_ACCESS_KEY"]} + run("docker", "run", "-d", "--name", metadata["container"], + "--restart", "unless-stopped", "--user", "10001:10001", + "--memory", "2g", "--cpus", "2", "--pids-limit", "256", + "--log-driver", "local", "--log-opt", "max-size=10m", + "--log-opt", "max-file=3", "-p", "127.0.0.1::9000", + "-v", metadata["volume"] + ":/data", + "-e", "RUSTFS_ACCESS_KEY", "-e", "RUSTFS_SECRET_KEY", + "-e", "RUSTFS_OBS_LOG_DIRECTORY=/data/logs", + metadata["provider_image"], "/data", env=provider_env) + endpoint = "http://" + run("docker", "port", metadata["container"], "9000/tcp") + credentials["AWS_ENDPOINT"] = endpoint + save(state / "secrets.json", credentials, private=True) + env = environment(state) + for _ in range(60): + try: + run("aws", "--endpoint-url", endpoint, "s3api", "head-bucket", + "--bucket", metadata["bucket"], env=env) + break + except subprocess.CalledProcessError: + try: + run("aws", "--endpoint-url", endpoint, "s3api", "create-bucket", + "--bucket", metadata["bucket"], env=env) + break + except subprocess.CalledProcessError: + time.sleep(1) + else: + raise RuntimeError("RustFS did not become ready; inspect its container logs") + with (state / "server.log").open("ab") as output: + process = subprocess.Popen([str(binary), str(state / "config.json")], + env=env, stdout=output, stderr=output, + stdin=subprocess.DEVNULL, start_new_session=True) + (state / "server.pid").write_text(str(process.pid)) + for _ in range(150): + if process.poll() is not None: + raise RuntimeError("Canopy exited; inspect server.log") + if ready(metadata["url"]): + return + time.sleep(0.2) + # Startup remains supervised by the running node; don't silently kill it. + raise RuntimeError("Canopy readiness timed out; inspect server.log before retrying") + + +def stop(state, metadata): + pid = owned_pid(state, metadata) + if pid: + os.kill(pid, signal.SIGTERM) + for _ in range(600): + if not owned_pid(state, metadata): + break + time.sleep(0.2) + else: + raise RuntimeError("node is still draining; provider left running") + run("docker", "stop", "--time", "30", metadata["container"]) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("action", choices=("start", "stop", "status")) + parser.add_argument("--state-dir", required=True, type=Path) + parser.add_argument("--binary", type=Path) + parser.add_argument("--port", type=int, default=18080) + parser.add_argument("--disk-gib", type=int, default=64) + parser.add_argument("--active-repositories", type=int, default=3) + parser.add_argument("--provider-image", default=PROVIDER_IMAGE) + args = parser.parse_args() + if not 1 <= args.port <= 65535 or args.disk_gib < 1 or not 1 <= args.active_repositories <= 9999: + parser.error("port must be 1..65535, disk-gib positive, and active-repositories 1..9999") + args.state_dir = args.state_dir.resolve() + if args.action == "start": + args.state_dir.mkdir(parents=True, exist_ok=True, mode=0o700) + metadata = initialize(args) + start(args, metadata) + else: + metadata = json.loads((args.state_dir / "deployment.json").read_text()) + if args.action == "stop": + stop(args.state_dir, metadata) + print(json.dumps({"url": metadata["url"], "ready": ready(metadata["url"]), + "pid": owned_pid(args.state_dir, metadata), + "state_dir": str(args.state_dir), + "credentials_file": str(args.state_dir / "secrets.json")}, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/test_local_eval.py b/scripts/test_local_eval.py new file mode 100644 index 0000000..5fd9be6 --- /dev/null +++ b/scripts/test_local_eval.py @@ -0,0 +1,63 @@ +"""Focused guards for persistent evaluation identity and secret handling.""" +import argparse +import os +from pathlib import Path +import tempfile +import unittest +from unittest.mock import patch + +import local_eval + + +class EvaluationIdentityTests(unittest.TestCase): + def test_restart_reuses_identity_and_private_credentials(self): + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + binary = root / "canopy" + binary.write_bytes(b"release fixture") + state = root / "state" + state.mkdir() + args = argparse.Namespace(state_dir=state, binary=binary, port=18080, + provider_image="fixture@sha256:abc", disk_gib=64, active_repositories=3) + first = local_eval.initialize(args) + original = (state / "secrets.json").read_bytes() + config = (state / "config.json").read_bytes() + args.port = 18081 + self.assertEqual(first, local_eval.initialize(args)) + self.assertEqual(original, (state / "secrets.json").read_bytes()) + self.assertEqual(config, (state / "config.json").read_bytes()) + if os.name == "posix": + self.assertEqual(0o600, (state / "secrets.json").stat().st_mode & 0o777) + self.assertNotIn(b"CANOPY_GIT_TOKEN", config) + with patch.dict(os.environ, {"AWS_SESSION_TOKEN": "unrelated-token", + "AWS_ENDPOINT": "unrelated-endpoint"}): + environment = local_eval.environment(state) + self.assertNotIn("AWS_SESSION_TOKEN", environment) + self.assertNotIn("AWS_ENDPOINT", environment) + + def test_existing_unrelated_directory_is_rejected(self): + with tempfile.TemporaryDirectory() as temporary: + state = Path(temporary) + (state / "unrelated").write_text("preserve me") + with self.assertRaisesRegex(RuntimeError, "not empty"): + local_eval.initialize(argparse.Namespace(state_dir=state)) + + def test_stale_pid_does_not_signal_an_unrelated_process(self): + with tempfile.TemporaryDirectory() as temporary: + state = Path(temporary) + (state / "server.pid").write_text("123") + with patch.object(local_eval, "run", return_value="another-service"): + self.assertIsNone(local_eval.owned_pid(state, {"binary": "/test/canopy"})) + + def test_changed_binary_is_rejected_before_provider_operations(self): + with tempfile.TemporaryDirectory() as temporary: + state = Path(temporary) + binary = state / "canopy" + binary.write_bytes(b"changed release") + with self.assertRaisesRegex(RuntimeError, "binary changed"): + local_eval.start(argparse.Namespace(state_dir=state), + {"binary": str(binary), "binary_sha256": "old"}) + + +if __name__ == "__main__": + unittest.main() diff --git a/src/deployment/backup/bodies.rs b/src/deployment/backup/bodies.rs index db7af14..c434136 100644 --- a/src/deployment/backup/bodies.rs +++ b/src/deployment/backup/bodies.rs @@ -10,6 +10,8 @@ use object_store::{ObjectStore, prefix::PrefixStore}; #[derive(Clone, Copy)] enum BodyKind { Git, + Pack, + PackIndex, Lfs, } enum Reference { @@ -93,7 +95,12 @@ impl Deployment { )); } } - for kind in [BodyKind::Git, BodyKind::Lfs] { + for kind in [ + BodyKind::Git, + BodyKind::Pack, + BodyKind::PackIndex, + BodyKind::Lfs, + ] { let mut cursor = Vec::new(); loop { let page = read_page(database.clone(), kind, cursor).await?; @@ -203,6 +210,8 @@ async fn read_page( let connection = Connection::open_with_flags(database, OpenFlags::SQLITE_OPEN_READ_ONLY | OpenFlags::SQLITE_OPEN_NO_MUTEX)?; let sql = match kind { BodyKind::Git => "SELECT oid, size, digest, external_sha256 FROM objects WHERE storage = 'external' AND oid > ?1 ORDER BY oid LIMIT 256", + BodyKind::Pack => "SELECT DISTINCT pack_oid, pack_size, pack_digest, sha256 FROM git_packs WHERE pack_oid > ?1 ORDER BY pack_oid LIMIT 256", + BodyKind::PackIndex => "SELECT DISTINCT index_oid, index_size, index_digest, index_sha256 FROM git_packs WHERE index_oid > ?1 ORDER BY index_oid LIMIT 256", BodyKind::Lfs => "SELECT sha256, size, digest, sha256 FROM lfs_objects WHERE sha256 > ?1 ORDER BY sha256 LIMIT 256", }; let mut statement = connection.prepare(sql)?; @@ -217,7 +226,7 @@ async fn read_page( let digest = digest.try_into().map_err(|_| BackupError::Invalid("invalid body digest"))?; let sha256 = sha256.try_into().map_err(|_| BackupError::Invalid("invalid body SHA-256"))?; page.push(match kind { - BodyKind::Git => Reference::Git(LargeBlobReference { oid: oid.try_into().map_err(|_| BackupError::Invalid("invalid Git OID"))?, size, blake3: digest, sha256 }), + BodyKind::Git | BodyKind::Pack | BodyKind::PackIndex => Reference::Git(LargeBlobReference { oid: oid.try_into().map_err(|_| BackupError::Invalid("invalid Git OID"))?, size, blake3: digest, sha256 }), BodyKind::Lfs => Reference::Lfs(LfsObject { sha256, size, parts_digest: digest }), }); } diff --git a/src/git_cache.rs b/src/git_cache.rs index 1a0ba85..a32adb4 100644 --- a/src/git_cache.rs +++ b/src/git_cache.rs @@ -1,11 +1,11 @@ //! Disposable Git files charged to the node's shared disk budget. use std::{ - collections::BTreeMap, + collections::{BTreeMap, BTreeSet, HashSet}, fs::{self, File, OpenOptions}, io::{self, BufWriter, Write}, path::{Path, PathBuf}, - sync::{Arc, OnceLock}, + sync::{Arc, OnceLock, RwLock}, }; use cellule_ltx::{DiskBudget, DiskReservation, LtxError}; @@ -55,6 +55,14 @@ pub(crate) struct GitCache { // Only durable hydration writes this cache. Stripe by OID so concurrent // fetches share a completed loose object without serializing all objects. object_writes: OnceLock<[Arc>; 64]>, + packed: RwLock>, + durable_packs: RwLock>, + pub(crate) selection: Mutex<()>, + pub(crate) prepared: Mutex>, + pub(crate) loose_objects: std::sync::atomic::AtomicU64, + pub(crate) pack_files: std::sync::atomic::AtomicU64, + pub(crate) hydrating: std::sync::atomic::AtomicU64, + pub(crate) write_generation: std::sync::atomic::AtomicU64, } impl GitCache { @@ -87,6 +95,14 @@ impl GitCache { reservation: Some(budget.try_reserve(0)?), objects, object_writes: OnceLock::new(), + packed: RwLock::new(HashSet::new()), + durable_packs: RwLock::new(HashSet::new()), + selection: Mutex::new(()), + prepared: Mutex::new(BTreeSet::new()), + loose_objects: std::sync::atomic::AtomicU64::new(0), + pack_files: std::sync::atomic::AtomicU64::new(0), + hydrating: std::sync::atomic::AtomicU64::new(0), + write_generation: std::sync::atomic::AtomicU64::new(0), }); for directory in ["objects/info", "objects/pack", "refs/heads", "refs/tags", "hooks"] { fs::create_dir_all(cache.git_dir().join(directory))?; @@ -119,6 +135,18 @@ impl GitCache { self.root().join("repo.git") } + pub(crate) fn object_cache(self: &Arc) -> Arc { + self.objects + .as_ref() + .map_or_else(|| Arc::clone(self), Arc::clone) + } + + pub(crate) fn hydration_guard(self: &Arc) -> HydrationGuard { + self.hydrating + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + HydrationGuard(Arc::clone(self)) + } + fn reservation(&self) -> io::Result<&DiskReservation> { self.reservation .as_ref() @@ -170,6 +198,14 @@ impl GitCache { } fn object_present(&self, oid: crate::ObjectId) -> io::Result { + if self + .packed + .read() + .map_err(|_| io::Error::other("packed inventory poisoned"))? + .contains(&oid) + { + return Ok(true); + } match fs::symlink_metadata(self.object_path(oid)) { Ok(metadata) if metadata.is_file() => Ok(true), Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(false), @@ -269,6 +305,17 @@ impl GitCache { temporary .persist_noclobber(destination) .map_err(|error| error.error)?; + cache + .packed + .write() + .map_err(|_| io::Error::other("verified inventory poisoned"))? + .insert(oid); + cache + .loose_objects + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); + cache + .write_generation + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); Ok(()) }) .await? @@ -276,24 +323,33 @@ impl GitCache { pub(crate) async fn store_blob( self: &Arc, - mut reader: LargeBlobRead, + reader: LargeBlobRead, ) -> Result<(), CacheError> { - let reference = reader.reference(); - let write = self.object_write_lock(reference.oid).lock_owned().await; + self.store_blob_reader(BlobReader::External(reader)).await + } + pub(crate) async fn store_native_blob( + self: &Arc, + reader: crate::pack_store::NativePackedRead, + ) -> Result<(), CacheError> { + self.store_blob_reader(BlobReader::Packed(reader)).await + } + async fn store_blob_reader(self: &Arc, mut reader: BlobReader) -> Result<(), CacheError> { + let (oid, size) = reader.metadata(); + let write = self.object_write_lock(oid).lock_owned().await; let cache = Arc::clone(self); let pending = tokio::task::spawn_blocking(move || { - if reference.oid.format() != cache.object_format { + if oid.format() != cache.object_format { return Err(io::Error::new( io::ErrorKind::InvalidData, "Git object format mismatch", ) .into()); } - if cache.object_present(reference.oid)? { + if cache.object_present(oid)? { return Ok::<_, CacheError>(None); } - let (mut encoder, temporary, destination) = cache.object_writer(reference.oid)?; - encoder.write_all(format!("blob {}\0", reference.size).as_bytes())?; + let (mut encoder, temporary, destination) = cache.object_writer(oid)?; + encoder.write_all(format!("blob {}\0", size).as_bytes())?; Ok::<_, CacheError>(Some((encoder, temporary, destination))) }) .await??; @@ -309,12 +365,24 @@ impl GitCache { }) .await??; } + let cache = Arc::clone(self); tokio::task::spawn_blocking(move || { let _write = write; drop(encoder.finish()?); temporary .persist_noclobber(destination) .map_err(|error| error.error)?; + cache + .packed + .write() + .map_err(|_| io::Error::other("verified inventory poisoned"))? + .insert(oid); + cache + .loose_objects + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); + cache + .write_generation + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); Ok::<_, CacheError>(()) }) .await? @@ -360,6 +428,10 @@ impl GitCache { }) .await? } + + pub(crate) fn packed_count(&self) -> usize { + self.packed.read().expect("packed inventory poisoned").len() + } } impl Drop for GitCache { @@ -436,3 +508,36 @@ fn tree_bytes(path: &Path) -> io::Result { #[cfg(test)] #[path = "git_cache/tests.rs"] mod tests; + +mod maintenance; + +pub(crate) struct HydrationGuard(Arc); +impl Drop for HydrationGuard { + fn drop(&mut self) { + self.0 + .hydrating + .fetch_sub(1, std::sync::atomic::Ordering::SeqCst); + } +} + +enum BlobReader { + External(LargeBlobRead), + Packed(crate::pack_store::NativePackedRead), +} +impl BlobReader { + fn metadata(&self) -> (crate::ObjectId, u64) { + match self { + Self::External(read) => (read.reference().oid, read.reference().size), + Self::Packed(read) => (read.oid, read.size), + } + } + async fn next(&mut self) -> Result, CacheError> { + match self { + Self::External(read) => Ok(read.next().await?), + Self::Packed(read) => read + .next() + .await + .map_err(|error| io::Error::other(error).into()), + } + } +} diff --git a/src/git_cache/maintenance.rs b/src/git_cache/maintenance.rs new file mode 100644 index 0000000..c4cc823 --- /dev/null +++ b/src/git_cache/maintenance.rs @@ -0,0 +1,533 @@ +//! Immutable cache generations: never repack/delete files beneath an active reader. +use super::*; +use crate::git_http::{GitHttpError, GitProcess, WORKER_DEADLINE, read_bounded}; +use std::process::Stdio; +use tokio::io::AsyncWriteExt; + +fn worker_error(error: GitHttpError) -> CacheError { + io::Error::other(error).into() +} + +// Native receive-pack writes v2 indexes. Reject unsupported layouts rather than +// guessing an inventory and incorrectly skipping authoritative hydration. +pub(crate) fn index_ids( + path: &Path, + format: crate::ObjectFormat, +) -> io::Result> { + let data = fs::read(path)?; + if data.get(..8) != Some(b"\xfftOc\0\0\0\x02") || data.len() < 1032 { + return Err(io::Error::other("unsupported Git pack index")); + } + let count = u32::from_be_bytes(data[1028..1032].try_into().unwrap()) as usize; + let width = if format == crate::ObjectFormat::Sha1 { + 20 + } else { + 32 + }; + let end = 1032_usize + .checked_add( + count + .checked_mul(width) + .ok_or_else(|| io::Error::other("pack index overflow"))?, + ) + .ok_or_else(|| io::Error::other("pack index overflow"))?; + let ids = data + .get(1032..end) + .ok_or_else(|| io::Error::other("truncated Git pack index"))?; + ids.chunks_exact(width) + .map(|id| { + id.try_into() + .map_err(|_| io::Error::other("invalid pack OID")) + }) + .collect() +} + +// Follow physical pack order to retain delta-base locality while verifying large +// histories. Hash order forces needless repeated decompression of distant bases. +fn index_order(path: &Path, format: crate::ObjectFormat) -> io::Result> { + let data = fs::read(path)?; + let ids = index_ids(path, format)?; + let count = ids.len(); + let width = format.bytes(); + let offsets = 1032_usize + .checked_add( + count + .checked_mul(width + 4) + .ok_or_else(|| io::Error::other("pack index overflow"))?, + ) + .ok_or_else(|| io::Error::other("pack index overflow"))?; + let large = offsets + .checked_add( + count + .checked_mul(4) + .ok_or_else(|| io::Error::other("pack index overflow"))?, + ) + .ok_or_else(|| io::Error::other("pack index overflow"))?; + let payload_end = data + .len() + .checked_sub(2 * width) + .ok_or_else(|| io::Error::other("truncated index"))?; + if large > payload_end { + return Err(io::Error::other("truncated offsets")); + } + let mut ordered = Vec::with_capacity(count); + for n in 0..count { + let at = offsets + n * 4; + let raw = u32::from_be_bytes(data[at..at + 4].try_into().unwrap()); + let offset = if raw & 0x8000_0000 == 0 { + u64::from(raw) + } else { + let at = large + .checked_add( + ((raw & 0x7fff_ffff) as usize) + .checked_mul(8) + .ok_or_else(|| io::Error::other("offset overflow"))?, + ) + .ok_or_else(|| io::Error::other("offset overflow"))?; + let end = at + .checked_add(8) + .filter(|end| *end <= payload_end) + .ok_or_else(|| io::Error::other("truncated large offset"))?; + u64::from_be_bytes(data[at..end].try_into().unwrap()) + }; + let oid = data[1032 + n * width..1032 + (n + 1) * width] + .try_into() + .map_err(|_| io::Error::other("invalid OID"))?; + ordered.push((offset, oid)); + } + ordered.sort_unstable(); + Ok(ordered.into_iter().map(|(_, oid)| oid).collect()) +} + +impl GitCache { + pub(crate) fn has_durable_pack(&self, sha: [u8; 32]) -> bool { + self.durable_packs + .read() + .expect("durable inventory poisoned") + .contains(&sha) + } + pub(crate) fn mark_durable_pack(&self, sha: [u8; 32]) { + self.durable_packs + .write() + .expect("durable inventory poisoned") + .insert(sha); + } + pub(crate) async fn pack_sources( + self: &Arc, + ) -> Result)>, CacheError> { + let cache = Arc::clone(self); + tokio::task::spawn_blocking(move || { + let mut sources = Vec::new(); + for entry in fs::read_dir(cache.git_dir().join("objects/pack"))? { + let index = entry?.path(); + if index.extension().and_then(|v| v.to_str()) != Some("idx") { + continue; + } + let name = index + .file_stem() + .and_then(|v| v.to_str()) + .and_then(|v| v.strip_prefix("pack-")) + .ok_or_else(|| io::Error::other("invalid pack filename"))?; + let hash = crate::ObjectId::from_hex(name) + .map_err(|_| io::Error::other("invalid pack hash"))?; + if hash.format() != cache.object_format { + return Err(io::Error::other("invalid pack format").into()); + } + let ids = index_order(&index, cache.object_format)?; + sources.push((hash, index.with_extension("pack"), index, ids)); + } + Ok(sources) + }) + .await? + } + pub(crate) async fn install_pack( + self: &Arc, + hash: crate::ObjectId, + pack: LargeBlobRead, + index: LargeBlobRead, + ) -> Result<(), CacheError> { + if hash.format() != self.object_format { + return Err(io::Error::other("pack format mismatch").into()); + } + self.write_artifact( + format!("objects/pack/pack-{}.pack", hex::encode(hash)), + pack, + ) + .await?; + self.write_artifact( + format!("objects/pack/pack-{}.idx", hex::encode(hash)), + index, + ) + .await?; + let cache = Arc::clone(self); + tokio::task::spawn_blocking(move || { + let path = cache + .git_dir() + .join(format!("objects/pack/pack-{}.idx", hex::encode(hash))); + let data = fs::read(&path)?; + let width = cache.object_format.bytes(); + if data.len() < width * 2 + || &data[data.len() - 2 * width..data.len() - width] != hash.as_ref() + { + return Err(io::Error::other("index/pack binding mismatch").into()); + } + let ids = index_ids(&path, cache.object_format)?; + cache + .packed + .write() + .map_err(|_| io::Error::other("verified inventory poisoned"))? + .extend(ids); + cache + .pack_files + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); + cache + .write_generation + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + Ok(()) + }) + .await? + } + async fn write_artifact( + self: &Arc, + relative: String, + mut reader: LargeBlobRead, + ) -> Result<(), CacheError> { + let destination = self.git_dir().join(relative); + // Retry a canceled pair without downloading/charging a second full pack. + // File names alone (possibly SHA-1) are insufficient: verify all hashes. + let expected = reader.reference(); + let existing = destination.clone(); + if tokio::task::spawn_blocking(move || -> io::Result { + use sha2::Digest; + use std::io::Read; + let mut file = match File::open(existing) { + Ok(file) => file, + Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(false), + Err(error) => return Err(error), + }; + if file.metadata()?.len() != expected.size { + return Err(io::Error::other("cached artifact size mismatch")); + } + let mut oid = crate::git_format::ObjectHasher::new( + expected.oid.format(), + ObjectKind::Blob, + expected.size, + ); + let mut sha = sha2::Sha256::new(); + let mut blake = blake3::Hasher::new(); + let mut buffer = vec![0; 8 << 20]; + loop { + let n = file.read(&mut buffer)?; + if n == 0 { + break; + } + oid.update(&buffer[..n]); + sha.update(&buffer[..n]); + blake.update(&buffer[..n]); + } + if oid.finalize() != expected.oid + || <[u8; 32]>::from(sha.finalize()) != expected.sha256 + || *blake.finalize().as_bytes() != expected.blake3 + { + return Err(io::Error::other("cached artifact digest mismatch")); + } + Ok(true) + }) + .await?? + { + return Ok(()); + } + let (file, temporary) = tempfile::NamedTempFile::new_in( + destination + .parent() + .ok_or_else(|| io::Error::other("missing pack directory"))?, + )? + .into_parts(); + let mut writer = CacheWriter { + file, + cache: Arc::clone(self), + }; + while let Some(bytes) = reader.next().await? { + writer = tokio::task::spawn_blocking(move || { + writer.write_all(&bytes)?; + Ok::<_, io::Error>(writer) + }) + .await??; + } + tokio::task::spawn_blocking(move || { + writer.flush()?; + writer.file.sync_all()?; + drop(writer); + if destination.exists() { + // Compare complete bytes, not merely Git's SHA-1 filename. + use std::io::Read; + let mut old = File::open(&destination)?; + let mut new = File::open(&temporary)?; + let mut a = vec![0; 1 << 20]; + let mut b = vec![0; 1 << 20]; + loop { + let n = old.read(&mut a)?; + let m = new.read(&mut b)?; + if n != m || a[..n] != b[..m] { + return Err(io::Error::other( + "cached pack differs from authoritative bytes", + )); + } + if n == 0 { + break; + } + } + return Ok(()); + } + temporary + .persist_noclobber(destination) + .map_err(|error| error.error)?; + Ok(()) + }) + .await??; + Ok(()) + } + /// Reuse only packs whose *every* object was verified and durably recorded + /// during this ingestion. Extra/unverified objects disable this optimization. + pub(crate) async fn retain_verified_packs( + self: &Arc, + source: Arc, + verified: HashSet, + ) -> Result { + let cache = Arc::clone(self); + tokio::task::spawn_blocking(move || { + let mut retained = 0; + for entry in fs::read_dir(source.git_dir().join("objects/pack"))? { + let path = entry?.path(); + if path.extension().and_then(|v| v.to_str()) != Some("idx") { + continue; + } + let ids = index_ids(&path, cache.object_format)?; + if !ids.is_subset(&verified) { + continue; + } + let pack = path.with_extension("pack"); + // Charge before native-cache file copies. Failure keeps the + // attempted bytes charged until reconciliation or cache drop. + let bytes = fs::metadata(&path)?.len() + fs::metadata(&pack)?.len(); + cache + .reservation()? + .try_grow(bytes) + .map_err(io::Error::other)?; + for input in [&pack, &path] { + let destination = cache.git_dir().join("objects/pack").join( + input + .file_name() + .ok_or_else(|| io::Error::other("missing pack name"))?, + ); + if !destination.exists() { + let temporary = + tempfile::NamedTempFile::new_in(destination.parent().unwrap())?; + fs::copy(input, temporary.path())?; + temporary + .persist_noclobber(&destination) + .map_err(|error| error.error)?; + } + } + retained += ids.len(); + cache + .write_generation + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + cache + .pack_files + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); + cache + .packed + .write() + .map_err(|_| io::Error::other("packed inventory poisoned"))? + .extend(ids); + } + // Caller serializes ingestion/hydration while this operation runs. + cache.reservation()?.resize(tree_bytes(cache.root())?)?; + Ok(retained) + }) + .await? + } + + /// Enumerate a captured cache into a new self-contained pack. Old objects + /// and packs are untouched; dropping the last old reader reclaims them. + pub(crate) async fn repacked( + self: &Arc, + root: PathBuf, + budget: DiskBudget, + ) -> Result, CacheError> { + let next = Self::create(root, budget, "refs/heads/main", self.object_format).await?; + // Native writes bypass CacheWriter. Reserve conservative scratch room + // before starting, then reconcile the completed generation. This is + // admission, not a hard OS disk quota (the deployment owns that quota). + let reserve = self + .bytes()? + .checked_add( + (self.packed_count() as u64) + .checked_mul(96) + .ok_or_else(|| io::Error::other("maintenance index estimate overflow"))?, + ) + .and_then(|n| n.checked_add(64 << 20)) + .ok_or_else(|| io::Error::other("maintenance disk estimate overflow"))?; + next.reservation()? + .try_grow(reserve) + .map_err(io::Error::other)?; + let coverage = self.prepared.lock().await.clone(); + let durable = self + .durable_packs + .read() + .map_err(|_| io::Error::other("durable inventory poisoned"))? + .clone(); + let run = async { + let mut listing_command = crate::native_git::command(&self.git_dir())?; + listing_command + .args([ + "cat-file", + "--batch-all-objects", + "--batch-check=%(objectname)", + ]) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + let mut listing = GitProcess::spawn(listing_command, Arc::clone(self))?; + let mut pack_command = crate::native_git::command(&self.git_dir())?; + pack_command + .args([ + "pack-objects", + "--index-version=2", + "--delta-base-offset", + "--threads=1", + ]) + .arg(next.git_dir().join("objects/pack/pack")) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + let mut packing = + GitProcess::spawn(pack_command, (Arc::clone(self), Arc::clone(&next)))?; + let mut input = packing + .child + .stdin + .take() + .ok_or(GitHttpError::Interrupted)?; + let mut output = listing + .child + .stdout + .take() + .ok_or(GitHttpError::Interrupted)?; + let (_, listing_stderr, hash, packing_stderr) = tokio::try_join!( + async { + tokio::io::copy(&mut output, &mut input).await?; + input.shutdown().await?; + drop(input); + Ok::<_, GitHttpError>(()) + }, + read_bounded( + listing + .child + .stderr + .take() + .ok_or(GitHttpError::Interrupted)?, + 64 << 10 + ), + read_bounded( + packing + .child + .stdout + .take() + .ok_or(GitHttpError::Interrupted)?, + 128 + ), + read_bounded( + packing + .child + .stderr + .take() + .ok_or(GitHttpError::Interrupted)?, + 64 << 10 + ), + )?; + finish(&mut listing, listing_stderr).await?; + finish(&mut packing, packing_stderr).await?; + let hash = std::str::from_utf8(&hash) + .map_err(|_| GitHttpError::MalformedCgi)? + .trim(); + let oid = crate::ObjectId::from_hex(hash).map_err(|_| GitHttpError::MalformedCgi)?; + if oid.format() != self.object_format { + return Err(GitHttpError::MalformedCgi); + } + let pack = next + .git_dir() + .join(format!("objects/pack/pack-{hash}.pack")); + let mut command = crate::native_git::command(&next.git_dir())?; + command + .args(["index-pack", "--verify"]) + .arg(&pack) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + let mut verify = GitProcess::spawn(command, Arc::clone(&next))?; + let (_, stderr) = tokio::try_join!( + read_bounded( + verify + .child + .stdout + .take() + .ok_or(GitHttpError::Interrupted)?, + 128 + ), + read_bounded( + verify + .child + .stderr + .take() + .ok_or(GitHttpError::Interrupted)?, + 64 << 10 + ) + )?; + let status = verify.child.wait().await?; + verify.disarm(); + if !status.success() { + return Err(GitHttpError::GitExit { + status, + stderr: String::from_utf8_lossy(&stderr).into_owned(), + }); + } + let next_copy = Arc::clone(&next); + tokio::task::spawn_blocking(move || { + let ids = index_ids(&pack.with_extension("idx"), next_copy.object_format)?; + *next_copy + .packed + .write() + .map_err(|_| io::Error::other("packed inventory poisoned"))? = ids; + Ok::<_, io::Error>(()) + }) + .await + .map_err(|error| io::Error::other(error))??; + Ok::<_, GitHttpError>(()) + }; + let result = tokio::time::timeout(WORKER_DEADLINE, run) + .await + .map_err(|_| GitHttpError::Timeout) + .and_then(|v| v); + next.reconcile().await?; + result.map_err(worker_error)?; + next.pack_files + .store(1, std::sync::atomic::Ordering::Relaxed); + *next.prepared.lock().await = coverage; + *next + .durable_packs + .write() + .map_err(|_| io::Error::other("durable inventory poisoned"))? = durable; + Ok(next) + } +} + +async fn finish(process: &mut GitProcess, stderr: Vec) -> Result<(), GitHttpError> { + let status = process.child.wait().await?; + process.disarm(); + if !status.success() { + return Err(GitHttpError::GitExit { + status, + stderr: String::from_utf8_lossy(&stderr).into_owned(), + }); + } + Ok(()) +} diff --git a/src/git_cache/tests.rs b/src/git_cache/tests.rs index 09c775c..6c411ea 100644 --- a/src/git_cache/tests.rs +++ b/src/git_cache/tests.rs @@ -1,5 +1,68 @@ use super::*; +#[tokio::test] +async fn repacking_rotates_a_complete_cache_without_invalidating_active_readers() +-> Result<(), Box> { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let cache = GitCache::create( + root.path().into(), + budget.clone(), + "refs/heads/main", + crate::ObjectFormat::Sha1, + ) + .await?; + let mut ids = Vec::new(); + for n in 0..64 { + let body = format!("{n}\n{}", "shared historical contents\n".repeat(400)); + let oid = object_id(crate::ObjectFormat::Sha1, ObjectKind::Blob, body.as_bytes()); + cache + .store_object(oid, ObjectKind::Blob, body.into_bytes()) + .await?; + ids.push(oid); + } + let reader = Arc::clone(&cache); + let packed = cache.repacked(root.path().into(), budget.clone()).await?; + assert!(packed.missing_objects(ids.clone()).await?.is_empty()); + assert!(!packed.object_path(ids[0]).exists()); + assert_eq!(packed.packed_count(), 64); + let reused = GitCache::create( + root.path().into(), + budget.clone(), + "refs/heads/main", + crate::ObjectFormat::Sha1, + ) + .await?; + assert_eq!( + reused + .retain_verified_packs(Arc::clone(&packed), ids[..63].iter().copied().collect()) + .await?, + 0 + ); + assert_eq!( + reused + .retain_verified_packs(Arc::clone(&packed), ids.iter().copied().collect()) + .await?, + 64 + ); + assert!(reused.missing_objects(ids.clone()).await?.is_empty()); + drop(reused); + for source in [&reader, &packed] { + let output = crate::native_git::command(&source.git_dir())? + .args(["cat-file", "blob", &hex::encode(ids[0])]) + .output() + .await?; + assert!(output.status.success()); + assert!(output.stdout.starts_with(b"0\nshared historical contents")); + } + drop(cache); + drop(packed); + assert!(reader.object_path(ids[0]).is_file()); + drop(reader); + assert_eq!(budget.used(), 0); + Ok(()) +} + #[tokio::test] async fn concurrent_hydration_publishes_each_object_once() -> Result<(), Box> { @@ -407,3 +470,94 @@ async fn corrupt_stream_never_installs_a_reusable_object() -> Result<(), Box Result<(), Box> { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let source = GitCache::create( + root.path().into(), + budget.clone(), + "refs/heads/main", + crate::ObjectFormat::Sha1, + ) + .await?; + let body = vec![b'x'; 2 << 20]; + let oid = object_id(crate::ObjectFormat::Sha1, ObjectKind::Blob, &body); + let digest = *blake3::hash(&body).as_bytes(); + source + .store_object(oid, ObjectKind::Blob, body.clone()) + .await?; + let foreign = object_id(crate::ObjectFormat::Sha1, ObjectKind::Blob, b"foreign"); + source + .store_object(foreign, ObjectKind::Blob, b"foreign".to_vec()) + .await?; + let packed = source.repacked(root.path().into(), budget.clone()).await?; + let (hash, pack, index, ids) = packed.pack_sources().await?.pop().unwrap(); + assert_eq!( + ids.iter().copied().collect::>(), + [oid, foreign].into_iter().collect() + ); + let store = Arc::new(object_store::memory::InMemory::new()); + let reader = crate::pack_store::PackReader::new( + store, + [4; 16], + root.path().into(), + budget.clone(), + crate::ObjectFormat::Sha1, + ); + let record = crate::pack_store::PackRecord { + hash, + pack: reader.upload(pack).await?, + index: reader.upload(index).await?, + approved: false, + covered_through: 0, + }; + let target = GitCache::create( + root.path().into(), + budget.clone(), + "refs/heads/main", + crate::ObjectFormat::Sha1, + ) + .await?; + target + .store_native_blob( + reader + .native_reader(record.clone(), oid, body.len() as u64, digest) + .await?, + ) + .await?; + assert!(target.missing_objects(vec![oid]).await?.is_empty()); + assert_eq!(target.missing_objects(vec![foreign]).await?, vec![foreign]); + let output = crate::native_git::command(&target.git_dir())? + .args(["cat-file", "blob", &hex::encode(oid)]) + .output() + .await?; + assert!(output.status.success()); + assert_eq!(output.stdout, body); + assert!( + reader + .cache() + .await? + .missing_objects(vec![oid, foreign]) + .await? + .len() + == 2 + ); + let invalid = GitCache::create( + root.path().into(), + budget.clone(), + "refs/heads/main", + crate::ObjectFormat::Sha1, + ) + .await?; + assert!( + invalid + .store_native_blob(reader.native_reader(record, oid, 2 << 20, [0; 32]).await?) + .await + .is_err() + ); + assert_eq!(invalid.missing_objects(vec![oid]).await?, vec![oid]); + Ok(()) +} diff --git a/src/git_gateway.rs b/src/git_gateway.rs index 5c021e6..1952ff0 100644 --- a/src/git_gateway.rs +++ b/src/git_gateway.rs @@ -3,7 +3,7 @@ use crate::ReadIdentity; use std::{ - collections::{BTreeMap, BTreeSet}, + collections::{BTreeMap, BTreeSet, HashSet}, error::Error as StdError, path::{Path, PathBuf}, sync::Arc, @@ -37,6 +37,7 @@ mod candidates; mod discovery; mod fetch; mod hydration; +mod maintenance; mod push; mod ssh; @@ -103,6 +104,7 @@ pub struct GitGateway { signer_directory: Option>, certificate_seed: OnceCell<[u8; 32]>, large_blobs: LargeBlobStore, + pack_reader: Arc, lfs: LfsService, scratch_root: PathBuf, disk_budget: DiskBudget, @@ -119,12 +121,22 @@ impl GitGateway { disk_budget: DiskBudget, ) -> Self { let large_blobs = LargeBlobStore::new(Arc::clone(&blob_store), repository.repository_id()); + let pack_reader = Arc::clone(repository.pack_reader.get_or_init(|| { + Arc::new(crate::pack_store::PackReader::new( + Arc::clone(&blob_store), + repository.repository_id(), + scratch_root.clone(), + disk_budget.clone(), + repository.object_format(), + )) + })); let lfs = LfsService::new(Arc::clone(&repository), blob_store); Self { repository, signer_directory: None, certificate_seed: OnceCell::new(), large_blobs, + pack_reader, lfs, scratch_root, disk_budget, @@ -343,17 +355,12 @@ impl GitGateway { let mut objects = self.objects.lock().await; if objects.is_none() { *objects = Some(CachedObjects { - cache: GitCache::create( - self.scratch_root.clone(), - self.disk_budget.clone(), - &snapshot.head, - self.repository.object_format(), - ) - .await?, + cache: self.pack_reader.cache().await?, through: 0, }); } let shared = objects.as_mut().ok_or(GatewayError::MalformedCache)?; + self.restore_packs(shared).await?; if include_blobs { self.hydrate(shared).await?; } else { @@ -439,11 +446,44 @@ impl GitGateway { // Published refs already have durable graph closure. Excluding them avoids // re-reading old history; the final Cell transaction still verifies every new tip. let excluded = before.values().filter_map(|state| state.oid).collect(); - let mut objects = GitObjects::start(&backend.git_dir(), included, excluded)?; + let started = std::time::Instant::now(); + let initial_high_water = self + .repository + .object_high_water() + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))? + .output; + let mut sources = backend.cache.pack_sources().await?; + let mut archive = None; + let mut packed_ids = None; + if sources.len() == 1 { + let (hash, pack, index, ids) = sources.pop().ok_or(GatewayError::MalformedCache)?; + let record = crate::pack_store::PackRecord { + hash, + pack: self.pack_reader.upload(pack).await?, + index: self.pack_reader.upload(index).await?, + approved: false, + covered_through: 0, + }; + self.repository + .register_pack(new_identity()?, &record) + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))?; + archive = Some(record); + packed_ids = Some(ids); + } + let mut objects = if let Some(ids) = packed_ids { + GitObjects::packed(&backend.git_dir(), ids)? + } else { + GitObjects::start(&backend.git_dir(), included, excluded)? + }; + let mut batch = ObjectBatch::default(); + let mut verified = HashSet::new(); + let mut logged = std::time::Instant::now(); loop { - let mut candidates = Vec::with_capacity(MAX_OBJECTS); - for _ in 0..MAX_OBJECTS { + let mut candidates = Vec::with_capacity(crate::object_batch::MAX_BATCH_OBJECTS); + for _ in 0..crate::object_batch::MAX_BATCH_OBJECTS { let Some(oid) = objects.next().await? else { break; }; @@ -454,43 +494,81 @@ impl GitGateway { } let present = self .repository - .existing_objects(&candidates) + .canonical_headers(&candidates) .await - .map_err(|error| GatewayError::Cell(Box::new(error)))? - .output; - for oid in candidates.into_iter().filter(|oid| !present.contains(oid)) { - let mut input = objects.read(oid).await?; - let object = - if input.kind == ObjectKind::Blob && input.size > INLINE_OBJECT_LIMIT as u64 { - let uploaded = self - .large_blobs - .put(oid, input.size, &mut input.reader) + .map_err(|error| GatewayError::Cell(Box::new(error)))?; + for oid in candidates { + if let Some((kind, size, digest)) = present.get(&oid) { + if archive.is_some() { + objects + .read(oid) + .await? + .verify(*kind, *size, *digest) .await?; - input.finish().await?; + verified.insert(oid); + } + continue; + } + let mut input = objects.read(oid).await?; + let object = if input.kind == ObjectKind::Blob && archive.is_some() { + let size = input.size; + let digest = input.fingerprint().await?; + StoredObject { + oid, + kind: ObjectKind::Blob, + storage: ObjectStorage::Packed { + size, + blake3: digest, + pack: archive + .as_ref() + .ok_or(GatewayError::MalformedCache)? + .pack + .sha256, + }, + } + } else if input.kind == ObjectKind::Blob && input.size > INLINE_OBJECT_LIMIT as u64 + { + let uploaded = self + .large_blobs + .put(oid, input.size, &mut input.reader) + .await?; + input.finish().await?; + StoredObject { + oid, + kind: ObjectKind::Blob, + storage: ObjectStorage::External { + size: uploaded.size, + blake3: uploaded.blake3, + sha256: uploaded.sha256, + }, + } + } else { + let (kind, body) = input.body().await?; + if body.len() > INLINE_OBJECT_LIMIT { + self.repository + .stage_object(new_identity()?, kind, &body) + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))? + } else if let Some(record) = + archive.as_ref().filter(|_| kind == ObjectKind::Blob) + { StoredObject { oid, - kind: ObjectKind::Blob, - storage: ObjectStorage::External { - size: uploaded.size, - blake3: uploaded.blake3, - sha256: uploaded.sha256, + kind, + storage: ObjectStorage::Packed { + size: body.len() as u64, + blake3: *blake3::hash(&body).as_bytes(), + pack: record.pack.sha256, }, } } else { - let (kind, body) = input.body().await?; - if body.len() > INLINE_OBJECT_LIMIT { - self.repository - .stage_object(new_identity()?, kind, &body) - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))? - } else { - StoredObject { - oid, - kind, - storage: ObjectStorage::Inline(body), - } + StoredObject { + oid, + kind, + storage: ObjectStorage::Inline(body), } - }; + } + }; if let Err(object) = batch.try_push(object) { self.repository .put_objects(new_identity()?, std::mem::take(&mut batch)) @@ -500,6 +578,15 @@ impl GitGateway { .try_push(object) .map_err(|_| GatewayError::MalformedCache)?; } + verified.insert(oid); + } + if logged.elapsed().as_secs() >= 10 { + tracing::info!( + objects = verified.len(), + elapsed_seconds = started.elapsed().as_secs_f64(), + "persisting Git objects" + ); + logged = std::time::Instant::now(); } } objects.finish().await?; @@ -509,6 +596,52 @@ impl GitGateway { .await .map_err(|error| GatewayError::Cell(Box::new(error)))?; } + if let Some(record) = &archive { + self.repository + .approve_pack(new_identity()?, record.pack.sha256, verified.len()) + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))?; + } + let mut shared = self.objects.lock().await; + if let Some(shared) = shared.as_mut() { + let count = verified.len(); + match shared + .cache + .retain_verified_packs(Arc::clone(&backend.cache), verified) + .await + { + Ok(retained) if retained == count && shared.through == initial_high_water => { + if let Some(record) = &archive { + shared.cache.mark_durable_pack(record.pack.sha256); + } + shared.through = self + .repository + .object_high_water() + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))? + .output; + tracing::info!( + objects = retained, + through = shared.through, + "retained verified receive pack for immediate fetch" + ); + } + Ok(retained) => { + if retained == count { + if let Some(record) = &archive { + shared.cache.mark_durable_pack(record.pack.sha256); + } + } + } + Err(error) => { + tracing::warn!(error = ?error, "receive pack cache reuse skipped; durable hydration remains available") + } + } + } + tracing::info!( + elapsed_seconds = started.elapsed().as_secs_f64(), + "persisted Git objects" + ); Ok(()) } } diff --git a/src/git_gateway/fetch.rs b/src/git_gateway/fetch.rs index 13910f7..fa1a060 100644 --- a/src/git_gateway/fetch.rs +++ b/src/git_gateway/fetch.rs @@ -116,18 +116,114 @@ impl GitGateway { // The shared object cache publishes loose objects atomically and // coordinates duplicate OID writes. Hold the gateway lock only long // enough to borrow it; slow fetches must not queue behind each other. - let shared = { + let shared = cached.backend.cache.object_cache(); + let _hydrating = shared.hydration_guard(); + let through = { let objects = self.objects.lock().await; - Arc::clone(&objects.as_ref().ok_or(GatewayError::MalformedCache)?.cache) + objects + .as_ref() + .filter(|objects| Arc::ptr_eq(&objects.cache, &shared)) + .map_or(0, |objects| objects.through) }; + if self + .repository + .object_high_water() + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))? + .output + <= through + { + shared + .prepared + .lock() + .await + .extend(request.wants.iter().map(|oid| (*oid, true))); + return Ok(()); + } let roots: Vec<_> = request.wants.iter().copied().collect(); - self.hydrate_selected(&shared, request.wants).await?; let unfiltered = request.filter.is_none(); + if roots.iter().all(|oid| { + shared.prepared.try_lock().ok().is_some_and(|prepared| { + prepared.contains(&(*oid, true)) + || (request.filter.as_deref() == Some("blob:none") + && prepared.contains(&(*oid, false))) + }) + }) { + return Ok(()); + } + let _selection = shared.selection.lock().await; + // A concurrent cold request may have completed while we waited. + if (unfiltered || request.filter.as_deref() == Some("blob:none")) + && roots.iter().all(|oid| { + shared.prepared.try_lock().ok().is_some_and(|prepared| { + prepared.contains(&(*oid, true)) + || (!unfiltered && prepared.contains(&(*oid, false))) + }) + }) + { + return Ok(()); + } + self.hydrate_selected(&shared, request.wants).await?; // The certified Cell graph already names every reachable blob. A full // fetch can hydrate those bodies during the structural walk and avoid // a second native traversal over the same cold history. - self.hydrate_structure(&shared, &roots, unfiltered).await?; + self.hydrate_structure(&shared, &roots, unfiltered, through) + .await?; if unfiltered || request.filter.as_deref() == Some("blob:none") { + shared + .prepared + .lock() + .await + .extend(roots.iter().map(|oid| (*oid, unfiltered))); + if unfiltered && through == 0 { + // Count a covering OID index, not the large body table. A cache + // inventory consists exclusively of verified durable IDs. Equal + // cardinality therefore proves the entire captured Cell is warm. + let result = self + .repository + .sql + .query( + None, + SqlBatch { + statements: vec![ + SqlStatement { + sql: "SELECT COUNT(oid) FROM objects".into(), + parameters: vec![], + }, + SqlStatement { + sql: "SELECT COALESCE(MAX(sequence), 0) FROM objects".into(), + parameters: vec![], + }, + ], + }, + ) + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))?; + if let (Some([SqlValue::Integer(count)]), Some([SqlValue::Integer(high_water)])) = ( + result + .output + .first() + .and_then(|set| set.rows.first()) + .map(Vec::as_slice), + result + .output + .get(1) + .and_then(|set| set.rows.first()) + .map(Vec::as_slice), + ) { + if usize::try_from(*count).ok() == Some(shared.packed_count()) { + // Never invert build_cache's lock order by waiting here. + if let Ok(mut objects) = self.objects.try_lock() { + if let Some(objects) = objects + .as_mut() + .filter(|objects| Arc::ptr_eq(&objects.cache, &shared)) + { + objects.through = objects.through.max(*high_water); + } + } + } + } + } return Ok(()); } // Use the same native filter as upload-pack. Structure is present, so @@ -282,6 +378,7 @@ impl GitGateway { cache: &Arc, roots: &[crate::ObjectId], include_blobs: bool, + through: i64, ) -> Result<(), GatewayError> { let mut pending: BTreeSet<_> = roots.iter().copied().collect(); let mut visited = BTreeSet::new(); @@ -306,13 +403,14 @@ impl GitGateway { let mut parameters: Vec<_> = ids.iter().map(|oid| SqlValue::Blob(oid.to_vec())).collect(); parameters.extend([ + SqlValue::Integer(through), SqlValue::Blob(after_parent.clone()), SqlValue::Blob(after_parent.clone()), SqlValue::Blob(after_child.clone()), SqlValue::Integer(MAX_OBJECTS as i64), ]); let result = self.repository.sql.query(None, SqlBatch { statements: vec![SqlStatement { - sql: format!("SELECT e.parent, e.child, o.kind FROM object_edges e JOIN objects o ON o.oid = e.child WHERE e.parent IN ({placeholders}) {kind_filter} AND (e.parent > ? OR (e.parent = ? AND e.child > ?)) ORDER BY e.parent, e.child LIMIT ?"), + sql: format!("SELECT e.parent, e.child, o.kind FROM object_edges e JOIN objects o ON o.oid = e.child JOIN objects p ON p.oid = e.parent WHERE e.parent IN ({placeholders}) AND p.sequence > ? {kind_filter} AND (e.parent > ? OR (e.parent = ? AND e.child > ?)) ORDER BY e.parent, e.child LIMIT ?"), parameters, }] }).await.map_err(|error| GatewayError::Cell(Box::new(error)))?; let rows = &result diff --git a/src/git_gateway/hydration.rs b/src/git_gateway/hydration.rs index 8644214..b91bfe7 100644 --- a/src/git_gateway/hydration.rs +++ b/src/git_gateway/hydration.rs @@ -11,9 +11,38 @@ pub(super) struct Hydration { } impl GitGateway { + pub(super) async fn restore_packs( + &self, + shared: &mut CachedObjects, + ) -> Result<(), GatewayError> { + let mut after = Vec::new(); + loop { + let page = self + .repository + .approved_packs(&after) + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))?; + if page.is_empty() { + break; + } + for record in &page { + self.pack_reader.install(&shared.cache, record).await?; + shared.through = shared.through.max(record.covered_through); + } + after = page + .last() + .ok_or(GatewayError::MalformedCache)? + .pack + .sha256 + .to_vec(); + } + Ok(()) + } + pub(super) async fn hydrate(&self, shared: &mut CachedObjects) -> Result<(), GatewayError> { let started = Instant::now(); let cache = &shared.cache; + let _selection = cache.selection.lock().await; let cursor = &mut shared.through; let from_sequence = *cursor; // Bound this refresh even when other writers keep appending objects. @@ -79,6 +108,28 @@ impl GitGateway { let read = Instant::now(); let body = match object.storage { ObjectStorage::Inline(body) => body, + ObjectStorage::Packed { pack, size, blake3 } => { + let record = self + .repository + .pack_record(pack) + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))?; + if record.approved { + self.pack_reader.install(cache, &record).await?; + return Ok(None); + } + let reader = self + .pack_reader + .native_reader(record, object.oid, size, blake3) + .await?; + stats.body_time += read.elapsed(); + let written = Instant::now(); + cache.store_native_blob(reader).await?; + stats.cache_time += written.elapsed(); + stats.objects += 1; + stats.bytes += size; + return Ok(None); + } ObjectStorage::Chunked { upload, size, diff --git a/src/git_gateway/maintenance.rs b/src/git_gateway/maintenance.rs new file mode 100644 index 0000000..edc43ee --- /dev/null +++ b/src/git_gateway/maintenance.rs @@ -0,0 +1,59 @@ +use super::*; +use std::sync::atomic::Ordering; + +impl GitGateway { + /// Best-effort maintenance has separate process admission and never owns a + /// user transfer slot. A failed job leaves the previous generation serving. + pub(crate) async fn maintain(&self) -> Result<(), GatewayError> { + static JOBS: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(1); + let Ok(_job) = JOBS.try_acquire() else { + return Ok(()); + }; + let (old, through, generation) = { + let Ok(objects) = self.objects.try_lock() else { + return Ok(()); + }; + let Some(objects) = objects.as_ref() else { + return Ok(()); + }; + if objects.cache.hydrating.load(Ordering::SeqCst) > 0 { + return Ok(()); + } + if objects.cache.loose_objects.load(Ordering::Relaxed) < 1024 + && objects.cache.pack_files.load(Ordering::Relaxed) < 8 + { + return Ok(()); + } + ( + Arc::clone(&objects.cache), + objects.through, + objects.cache.write_generation.load(Ordering::SeqCst), + ) + }; + let started = std::time::Instant::now(); + let next = old + .repacked(self.scratch_root.clone(), self.disk_budget.clone()) + .await?; + // Same lock order as fetch_cache/build_cache. Foreground requests are + // never locked out while pack-objects runs. Changed inventories retry. + let mut refs = self.cache.lock().await; + let mut objects = self.objects.lock().await; + let Some(objects) = objects.as_mut() else { + return Ok(()); + }; + if !Arc::ptr_eq(&objects.cache, &old) + || objects.through != through + || old.hydrating.load(Ordering::SeqCst) > 0 + || old.write_generation.load(Ordering::SeqCst) != generation + { + return Ok(()); + } + tracing::info!(repository = %hex::encode(self.repository.repository_id()), + objects = next.packed_count(), previous_bytes = old.bytes()?, packed_bytes = next.bytes()?, + elapsed_seconds = started.elapsed().as_secs_f64(), "published background Git repack generation"); + self.pack_reader.replace(&old, Arc::clone(&next)).await; + objects.cache = next; + *refs = None; + Ok(()) + } +} diff --git a/src/git_http.rs b/src/git_http.rs index 4075c1d..0ff6924 100644 --- a/src/git_http.rs +++ b/src/git_http.rs @@ -30,6 +30,9 @@ const MAX_CGI_OUTPUT_BYTES: usize = 64 * 1024 * 1024; const MAX_CGI_STDERR_BYTES: usize = 64 * 1024; const CHUNK_BYTES: usize = 64 * 1024; const MAX_CGI_HEADER_BYTES: usize = 64 * 1024; +// Pack traversal and delta search can legitimately run for minutes before +// producing output. Cancellation still kills the entire process group. +pub(crate) const WORKER_DEADLINE: Duration = Duration::from_secs(3600); #[derive(Debug, thiserror::Error)] pub enum GitHttpError { @@ -243,7 +246,7 @@ impl GitHttpBackend { start_stream( process, (keep_alive, Arc::clone(&self.cache), request.body), - Duration::from_secs(120), + WORKER_DEADLINE, ) .await } @@ -350,6 +353,9 @@ async fn start_stream( .await .map_err(|_| GitHttpError::Timeout) .and_then(|result| result); + if let Err(error) = &result { + tracing::warn!(error = ?error, deadline_seconds = deadline.as_secs(), "Git response worker failed"); + } // Cleanup precedes any blocked delivery of the final error. drop(process); drop(head_sender); diff --git a/src/git_objects.rs b/src/git_objects.rs index e5e1871..96e0c45 100644 --- a/src/git_objects.rs +++ b/src/git_objects.rs @@ -167,7 +167,8 @@ impl GitObjectWalk { } pub(crate) struct GitObjects { - walk: GitObjectWalk, + walk: Option, + inventory: Option>, batch: Process, requests: ChildStdin, } @@ -181,14 +182,35 @@ impl GitObjects { let walk = GitObjectWalk::start(git_dir, included, excluded, false, None)?; let (batch, requests) = Process::start(git_dir, &["cat-file", "--batch"])?; Ok(Self { - walk, + walk: Some(walk), + inventory: None, + batch, + requests, + }) + } + + pub(crate) fn packed( + git_dir: &Path, + ids: Vec, + ) -> Result { + let (batch, requests) = Process::start(git_dir, &["cat-file", "--batch"])?; + Ok(Self { + walk: None, + inventory: Some(ids.into_iter()), batch, requests, }) } pub(crate) async fn next(&mut self) -> Result, ObjectReadError> { - self.walk.next().await + if let Some(inventory) = &mut self.inventory { + return Ok(inventory.next()); + } + self.walk + .as_mut() + .ok_or(ObjectReadError::Malformed)? + .next() + .await } pub(crate) async fn read( @@ -207,7 +229,9 @@ impl GitObjects { pub(crate) async fn finish(self) -> Result<(), ObjectReadError> { timeout(IO_TIMEOUT, async move { - self.walk.finish().await?; + if let Some(walk) = self.walk { + walk.finish().await?; + } drop(self.requests); let mut batch = self.batch; if header(&mut batch.output).await?.is_some() { @@ -247,6 +271,60 @@ pub(crate) struct GitObject<'a, R> { } impl GitObject<'_, R> { + /// Verify a packed body with constant memory, including oversized blobs. + pub(crate) async fn fingerprint(mut self) -> Result<[u8; 32], ObjectReadError> { + let expected = self.oid; + let mut canonical = + crate::git_format::ObjectHasher::new(expected.format(), self.kind, self.size); + let mut hash = blake3::Hasher::new(); + let mut buffer = vec![0; 64 << 10]; + loop { + let count = timeout(IO_TIMEOUT, self.reader.read(&mut buffer)) + .await + .map_err(|_| ObjectReadError::Timeout)??; + if count == 0 { + break; + } + canonical.update(&buffer[..count]); + hash.update(&buffer[..count]); + } + self.finish().await?; + if canonical.finalize() != expected { + return Err(ObjectReadError::Malformed); + } + Ok(*hash.finalize().as_bytes()) + } + + pub(crate) async fn verify( + mut self, + kind: ObjectKind, + size: u64, + digest: [u8; 32], + ) -> Result<(), ObjectReadError> { + if self.kind != kind || self.size != size { + return Err(ObjectReadError::Malformed); + } + let expected = self.oid; + let mut canonical = crate::git_format::ObjectHasher::new(expected.format(), kind, size); + let mut hash = blake3::Hasher::new(); + let mut buffer = vec![0; 64 << 10]; + loop { + let count = timeout(IO_TIMEOUT, self.reader.read(&mut buffer)) + .await + .map_err(|_| ObjectReadError::Timeout)??; + if count == 0 { + break; + } + canonical.update(&buffer[..count]); + hash.update(&buffer[..count]); + } + self.finish().await?; + if canonical.finalize() != expected || hash.finalize().as_bytes() != &digest { + return Err(ObjectReadError::Malformed); + } + Ok(()) + } + pub(crate) async fn body(mut self) -> Result<(ObjectKind, Vec), ObjectReadError> { let limit = if self.kind == ObjectKind::Blob { INLINE_OBJECT_LIMIT diff --git a/src/git_objects/tests.rs b/src/git_objects/tests.rs index f49285e..61d41f7 100644 --- a/src/git_objects/tests.rs +++ b/src/git_objects/tests.rs @@ -167,7 +167,7 @@ async fn dropping_reader_kills_both_children() -> TestResult { let tip = oid(directory.path(), "HEAD").await?; let objects = GitObjects::start(&directory.path().join(".git"), vec![tip], vec![])?; let pids = [ - objects.walk.process.child.id().unwrap(), + objects.walk.as_ref().unwrap().process.child.id().unwrap(), objects.batch.child.id().unwrap(), ]; drop(objects); diff --git a/src/graph.rs b/src/graph.rs index 0afd89f..1ab5c20 100644 --- a/src/graph.rs +++ b/src/graph.rs @@ -14,7 +14,7 @@ mod preparation; type Oid = crate::ObjectId; type Edge = (Oid, Option); -const MAX_CERTIFICATES: usize = 128; +const MAX_CERTIFICATES: usize = 512; #[derive(Default)] pub(crate) struct CertificateBatch(Vec); @@ -60,7 +60,7 @@ fn status_query(oids: &[Oid]) -> SqlBatch { SqlBatch { statements: vec![SqlStatement { sql: format!( - "SELECT o.oid, o.kind, CASE WHEN o.storage = 'external' THEN 0 ELSE o.size END, c.oid FROM objects o LEFT JOIN object_closure c ON c.oid = o.oid WHERE o.oid IN ({placeholders})" + "SELECT o.oid, o.kind, CASE WHEN o.storage IN ('external', 'packed') THEN 0 ELSE o.size END, c.oid FROM objects o LEFT JOIN object_closure c ON c.oid = o.oid WHERE o.oid IN ({placeholders})" ), parameters: oids .iter() @@ -106,7 +106,7 @@ pub(crate) struct CertifyObjects; impl Command for CertifyObjects { const MODULE: &'static str = RepositoryModule::NAME; const ID: u32 = 6; - const CODEC_VERSION: u32 = 2; + const CODEC_VERSION: u32 = 3; type Input = CertificateBatch; type Output = bool; @@ -249,7 +249,9 @@ fn object_edges( } // The immutable upload is verified before its SQLite record is published. // Blobs have no outgoing Git edges; no network I/O belongs in this transaction. - ("external", SqlValue::Null, SqlValue::Null) if kind == ObjectKind::Blob => Vec::new(), + ("external" | "packed", SqlValue::Null, SqlValue::Null) if kind == ObjectKind::Blob => { + Vec::new() + } ("chunked", SqlValue::Null, SqlValue::Blob(upload)) => { let invalid = || Error::Command("invalid stored graph object chunks"); let body = crate::object_chunks::body( diff --git a/src/graph/preparation.rs b/src/graph/preparation.rs index eb40a83..0d1d02d 100644 --- a/src/graph/preparation.rs +++ b/src/graph/preparation.rs @@ -26,8 +26,7 @@ enum PrepareError { } enum Visit { - Enter(Edge, Option), - Edges(std::vec::IntoIter), + Enter(Edge), Leave(Oid, ObjectKind, u64), } @@ -58,58 +57,141 @@ impl RepositoryCell { { return Ok(()); } + let started = std::time::Instant::now(); + // One preparer per process bounds aggregate memory while retaining no + // arbitrary repository/object-count ceiling. Durable certificates still + // recheck typed dependencies inside each bounded Cell transaction. + static PREPARERS: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(1); + let _preparer = PREPARERS + .acquire() + .await + .map_err(|_| Error::Command("graph admission closed"))?; + let mut nodes = HashMap::new(); + let mut ready = HashMap::new(); + let mut discovered = HashSet::new(); + let mut frontier: BTreeSet<_> = plan + .updates + .iter() + .filter_map(|update| update.new_oid) + .collect(); + while !frontier.is_empty() { + let page: Vec<_> = frontier.iter().take(MAX_CERTIFICATES).copied().collect(); + for oid in &page { + frontier.remove(oid); + discovered.insert(*oid); + } + let result = self.sql.query(None, status_query(&page)).await?; + let states = statuses(&result.output)?; + let mut structure = BTreeSet::new(); + for oid in &page { + let Some(state) = states.get(oid) else { + return Ok(()); + }; + if state.certified { + ready.insert(*oid, state.kind); + continue; + } + nodes.insert(*oid, (state.kind, state.bytes, Some(Vec::new()))); + if state.kind != ObjectKind::Blob { + structure.insert(*oid); + } + } + while !structure.is_empty() { + let ids: Vec<_> = structure + .iter() + .take(crate::object_batch::MAX_OBJECTS) + .copied() + .collect(); + let records = self.selected_objects(&ids).await?; + if records.is_empty() { + return Err(Error::Command("empty graph body page").into()); + } + for record in records { + structure.remove(&record.oid); + let oid = record.oid; + let kind = record.kind; + let body = match record.storage { + crate::ObjectStorage::Inline(body) => body, + crate::ObjectStorage::Chunked { .. } => { + self.object(oid, None) + .await? + .output + .ok_or(Error::Command("missing graph body"))? + .1 + } + _ => return Err(Error::Command("invalid structural storage").into()), + }; + let parsed = + tokio::task::spawn_blocking(move || edges(oid.format(), kind, &body)) + .await?; + if let Some(edges) = &parsed { + frontier.extend( + edges + .iter() + .map(|(oid, _)| *oid) + .filter(|oid| !discovered.contains(oid)), + ); + } + nodes + .get_mut(&oid) + .ok_or(Error::Command("missing graph node"))? + .2 = parsed; + } + } + } + drop(discovered); + let loaded = nodes.len(); let mut pending: Vec<_> = plan .updates .iter() .filter_map(|update| { update.new_oid.map(|oid| { - ( + Visit::Enter(( oid, update .name .starts_with("refs/heads/") .then_some(ObjectKind::Commit), - ) + )) }) }) - .collect::>() - .into_iter() - .map(|edge| Visit::Enter(edge, None)) .collect(); let mut visiting = HashSet::new(); - let mut ready = HashMap::new(); let mut batch = CertificateBatch::default(); let mut bytes = 0; while let Some(visit) = pending.pop() { - let (oid, expected, observed) = match visit { - Visit::Enter((oid, expected), observed) => (oid, expected, observed), - Visit::Edges(mut edges) => { - let page: Vec<_> = edges.by_ref().take(MAX_CERTIFICATES).collect(); - if page.is_empty() { + match visit { + Visit::Enter((oid, expected)) => { + if let Some(kind) = ready.get(&oid) { + if expected.is_some_and(|expected| expected != *kind) { + return Ok(()); + } continue; } - let oids: Vec<_> = page.iter().map(|(oid, _)| *oid).collect(); - let result = self.sql.query(None, status_query(&oids)).await?; - let states = statuses(&result.output)?; - pending.push(Visit::Edges(edges)); - for (oid, expected) in page.into_iter().rev() { - let Some(state) = states.get(&oid) else { - return Ok(()); - }; - if expected.is_some_and(|expected| expected != state.kind) { + // Dependencies outside the captured uncertified inventory + // must already be certified. The command checks this, and + // final publication always checks the roots again. + let Some((kind, weight, children)) = nodes.remove(&oid) else { + if visiting.contains(&oid) { return Ok(()); } - if !state.certified { - pending.push(Visit::Enter((oid, expected), Some(state.clone()))); - } + continue; + }; + if expected.is_some_and(|expected| expected != kind) || !visiting.insert(oid) { + return Ok(()); + } + let Some(children) = children else { + return Ok(()); + }; + pending.push(Visit::Leave(oid, kind, weight)); + for edge in children.into_iter().rev() { + pending.push(Visit::Enter(edge)); } - continue; } Visit::Leave(oid, kind, weight) => { if !batch.0.is_empty() && weight > VERIFY_BATCH_BYTES.saturating_sub(bytes) { self.certify_batch(std::mem::take(&mut batch)).await?; bytes = 0; - ready.clear(); } batch.0.push(oid); bytes += weight; @@ -118,54 +200,18 @@ impl RepositoryCell { if batch.0.len() == MAX_CERTIFICATES || bytes >= VERIFY_BATCH_BYTES { self.certify_batch(std::mem::take(&mut batch)).await?; bytes = 0; - ready.clear(); } - continue; - } - }; - if let Some(kind) = ready.get(&oid) { - if expected.is_some_and(|expected| expected != *kind) { - return Ok(()); } - continue; } - let state = if let Some(state) = observed { - state - } else { - let result = self.sql.query(None, status_query(&[oid])).await?; - let Some(state) = statuses(&result.output)?.remove(&oid) else { - return Ok(()); - }; - state - }; - if expected.is_some_and(|expected| expected != state.kind) { - return Ok(()); - } - if state.certified { - continue; - } - if !visiting.insert(oid) { - return Ok(()); - } - let edges = if state.kind == ObjectKind::Blob { - Vec::new() - } else { - let Some((kind, body)) = self.object(oid, None).await?.output else { - return Ok(()); - }; - let Some(edges) = - tokio::task::spawn_blocking(move || edges(oid.format(), kind, &body)).await? - else { - return Ok(()); - }; - edges - }; - pending.push(Visit::Leave(oid, state.kind, state.bytes)); - pending.push(Visit::Edges(edges.into_iter())); } if !batch.0.is_empty() { self.certify_batch(batch).await?; } + tracing::info!( + objects = loaded, + elapsed_seconds = started.elapsed().as_secs_f64(), + "prepared Git graph certificates" + ); Ok(()) } diff --git a/src/lib.rs b/src/lib.rs index 5a0dfc7..9af858c 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -40,6 +40,7 @@ mod native_git; mod object_batch; mod object_chunks; mod object_reads; +mod pack_store; pub mod pulls; mod push; mod refs; @@ -81,9 +82,9 @@ const COMMANDS: [OperationDescriptor; 9] = [ operation_with_codec(4, 6), OperationDescriptor { input_limit: object_batch::INPUT_LIMIT, - ..operation_with_codec(5, 3) + ..operation_with_codec(5, 5) }, - operation_with_codec(6, 2), + operation_with_codec(6, 3), operation_with_codec(7, 2), operation_with_codec(8, 2), operation_with_codec(9, 4), @@ -156,6 +157,11 @@ pub enum ObjectStorage { size: u64, blake3: [u8; 32], }, + Packed { + size: u64, + blake3: [u8; 32], + pack: [u8; 32], + }, External { size: u64, blake3: [u8; 32], @@ -201,6 +207,9 @@ impl CellModule for RepositoryModule { source.update(include_bytes!("object_chunks.rs")); source.update(include_bytes!("object_reads.rs")); source.update(include_bytes!("large_blob.rs")); + source.update(include_bytes!("pack_store.rs")); + source.update(include_bytes!("git_objects.rs")); + source.update(include_bytes!("git_gateway.rs")); source.update(include_bytes!("external.rs")); source.update(include_bytes!("push.rs")); source.update(include_bytes!("push/plan.rs")); @@ -293,6 +302,7 @@ pub struct RepositoryCell { sql: SqlCell, application: ApplicationHandle, target: CellTarget, + pack_reader: OnceLock>, } impl RepositoryCell { @@ -319,6 +329,7 @@ impl RepositoryCell { sql: application.sql::(target.clone())?, application: application.clone(), target, + pack_reader: OnceLock::new(), }) } @@ -350,7 +361,7 @@ impl RepositoryCell { SqlBatch { statements: vec![SqlStatement { sql: - "SELECT kind, body, digest, size, chunk_id FROM objects WHERE oid = ?1" + "SELECT kind, body, digest, size, chunk_id, storage, external_sha256 FROM objects WHERE oid = ?1" .into(), parameters: vec![SqlValue::Blob(oid.to_vec())], }], @@ -369,6 +380,8 @@ impl RepositoryCell { SqlValue::Blob(digest), SqlValue::Integer(size), upload, + SqlValue::Text(storage), + locator, ] = row.as_slice() else { return Err(cellule_runtime::InvocationError::NotStarted( @@ -386,6 +399,59 @@ impl RepositoryCell { )); } }; + if storage == "packed" { + let SqlValue::Blob(pack) = locator else { + return Err(cellule_runtime::InvocationError::NotStarted( + Error::Command("invalid pack locator"), + )); + }; + let pack: [u8; 32] = pack.as_slice().try_into().map_err(|_| { + cellule_runtime::InvocationError::NotStarted(Error::Command("invalid pack locator")) + })?; + if *size < 0 || *size > INLINE_OBJECT_LIMIT as i64 { + return Err(cellule_runtime::InvocationError::NotStarted( + Error::Command("object exceeds bounded body reader"), + )); + } + let record = self.pack_record(pack).await?; + let reader = + self.pack_reader + .get() + .ok_or(cellule_runtime::InvocationError::NotStarted( + Error::Command("packed reader unavailable"), + ))?; + let body = reader + .read_blob( + record, + oid, + *size as u64, + digest.as_slice().try_into().map_err(|_| { + cellule_runtime::InvocationError::NotStarted(Error::Command( + "invalid packed digest", + )) + })?, + ) + .await + .map_err(|error| { + cellule_runtime::InvocationError::NotStarted(Error::Facility { + name: "packed Git body", + source: Box::new(error), + }) + })?; + if kind != ObjectKind::Blob + || body.len() as i64 != *size + || object_id(oid.format(), kind, &body) != oid + || blake3::hash(&body).as_bytes() != digest.as_slice() + { + return Err(cellule_runtime::InvocationError::NotStarted( + Error::Command("corrupt packed Git body"), + )); + } + return Ok(Observed { + output: Some((kind, body)), + receipt: result.receipt, + }); + } let body = match (body, upload) { (SqlValue::Blob(bytes), SqlValue::Null) if usize::try_from(*size).ok() == Some(bytes.len()) => diff --git a/src/object_batch.rs b/src/object_batch.rs index 4466106..de6f4d2 100644 --- a/src/object_batch.rs +++ b/src/object_batch.rs @@ -15,6 +15,8 @@ use crate::{ }; pub(crate) const MAX_OBJECTS: usize = 128; +// Publication amortizes durable commits independently of bounded read pages. +pub(crate) const MAX_BATCH_OBJECTS: usize = 2048; pub(crate) const VERIFY_BATCH_BYTES: u64 = 64 * 1024 * 1024; pub(crate) const INPUT_LIMIT: u32 = 4 * 1024 * 1024; // Leave room for record metadata inside Cellule's bounded command wire format. @@ -33,14 +35,16 @@ impl ObjectBatch { pub fn try_push(&mut self, object: StoredObject) -> Result<(), StoredObject> { let bytes = match &object.storage { ObjectStorage::Inline(body) => body.len(), - ObjectStorage::External { .. } | ObjectStorage::Chunked { .. } => 0, + ObjectStorage::Packed { .. } + | ObjectStorage::External { .. } + | ObjectStorage::Chunked { .. } => 0, }; let verified = match &object.storage { ObjectStorage::Inline(body) => body.len() as u64, ObjectStorage::Chunked { size, .. } => *size, - ObjectStorage::External { .. } => 0, + ObjectStorage::External { .. } | ObjectStorage::Packed { .. } => 0, }; - if self.objects.len() == MAX_OBJECTS + if self.objects.len() == MAX_BATCH_OBJECTS || bytes > INLINE_OBJECT_LIMIT || bytes > INLINE_BATCH_BYTES - self.inline_bytes || i64::try_from(verified).is_err() @@ -90,6 +94,12 @@ impl WireValue for ObjectBatch { encoder.write_u64(*size)?; encoder.write_bytes(blake3)?; } + ObjectStorage::Packed { size, blake3, pack } => { + encoder.write_u8(3)?; + encoder.write_u64(*size)?; + encoder.write_bytes(blake3)?; + encoder.write_bytes(pack)?; + } ObjectStorage::External { size, blake3, @@ -107,7 +117,7 @@ impl WireValue for ObjectBatch { fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { let count = decoder.read_count()?; - if !(1..=MAX_OBJECTS).contains(&count) { + if !(1..=MAX_BATCH_OBJECTS).contains(&count) { return Err(CodecError::Invalid("object count is outside batch bounds")); } let mut batch = Self::default(); @@ -138,6 +148,11 @@ impl WireValue for ObjectBatch { blake3: fixed(decoder)?, sha256: fixed(decoder)?, }, + 3 => ObjectStorage::Packed { + size: decoder.read_u64()?, + blake3: fixed(decoder)?, + pack: fixed(decoder)?, + }, 2 => ObjectStorage::Chunked { upload: fixed(decoder)?, size: decoder.read_u64()?, @@ -165,7 +180,7 @@ pub(crate) struct PutObjects; impl Command for PutObjects { const MODULE: &'static str = RepositoryModule::NAME; const ID: u32 = 5; - const CODEC_VERSION: u32 = 3; + const CODEC_VERSION: u32 = 5; type Input = ObjectBatch; type Output = (); @@ -210,6 +225,28 @@ impl Command for PutObjects { SqlValue::Null, ) } + ObjectStorage::Packed { size, blake3, pack } => { + if object.kind != ObjectKind::Blob || i64::try_from(size).is_err() { + return Ok(CommandResult::Rejected(())); + } + let result = context.sql(&SqlBatch { + statements: vec![SqlStatement { + sql: "SELECT 1 FROM git_packs WHERE sha256 = ?1".into(), + parameters: vec![SqlValue::Blob(pack.to_vec())], + }], + })?; + if result.first().is_none_or(|set| set.rows.is_empty()) { + return Ok(CommandResult::Rejected(())); + } + ( + size as i64, + blake3.to_vec(), + "packed", + SqlValue::Null, + SqlValue::Blob(pack.to_vec()), + SqlValue::Null, + ) + } ObjectStorage::External { size, blake3, @@ -280,6 +317,17 @@ impl Command for PutObjects { if result.get(1).and_then(|set| set.rows.first()) != Some(&expected) { return Ok(CommandResult::Rejected(())); } + // Verified blobs are leaves. Certify them in this same savepoint, + // avoiding another full read/hash of every historical blob later. + if object.kind == ObjectKind::Blob { + context.sql(&SqlBatch { + statements: vec![SqlStatement { + sql: "INSERT INTO object_closure (oid) VALUES (?1) ON CONFLICT DO NOTHING" + .into(), + parameters: vec![SqlValue::Blob(object.oid.to_vec())], + }], + })?; + } } Ok(CommandResult::Success(())) } diff --git a/src/object_batch/tests.rs b/src/object_batch/tests.rs index 5cddd83..6b5bf95 100644 --- a/src/object_batch/tests.rs +++ b/src/object_batch/tests.rs @@ -11,7 +11,7 @@ fn inline(size: usize) -> StoredObject { #[test] fn maximum_batch_round_trips_below_the_operation_wire_limit() { let mut batch = ObjectBatch::default(); - for _ in 0..MAX_OBJECTS - 1 { + for _ in 0..MAX_BATCH_OBJECTS - 1 { assert!( batch .try_push(StoredObject { @@ -52,7 +52,7 @@ fn a_full_batch_returns_the_unconsumed_record() { }; let mut next = ObjectBatch::default(); assert!(next.try_push(leftover).is_ok()); - for _ in 1..MAX_OBJECTS { + for _ in 1..MAX_BATCH_OBJECTS { assert!(next.try_push(inline(0)).is_ok()); } assert!(next.try_push(inline(0)).is_err()); @@ -60,7 +60,7 @@ fn a_full_batch_returns_the_unconsumed_record() { #[test] fn decoding_enforces_aggregate_bytes_and_count_before_building_a_batch() { - for count in [0, MAX_OBJECTS + 1] { + for count in [0, MAX_BATCH_OBJECTS + 1] { let mut encoder = BoundedEncoder::new(INPUT_LIMIT).unwrap(); encoder.write_count(count).unwrap(); let bytes = encoder.finish(); @@ -138,3 +138,32 @@ fn chunk_references_bound_total_verification_bytes_and_round_trip() { let mut oversized = ObjectBatch::default(); assert!(oversized.try_push(chunked(u64::MAX)).is_err()); } + +#[test] +fn packed_metadata_batch_preserves_immutable_locator_without_blob_expansion() { + let mut batch = ObjectBatch::default(); + for _ in 0..MAX_BATCH_OBJECTS { + batch + .try_push(StoredObject { + oid: crate::ObjectId::Sha1([1; 20]), + kind: ObjectKind::Blob, + storage: ObjectStorage::Packed { + size: INLINE_OBJECT_LIMIT as u64, + blake3: [2; 32], + pack: [3; 32], + }, + }) + .ok() + .unwrap(); + } + let mut encoder = BoundedEncoder::new(INPUT_LIMIT).unwrap(); + batch.encode(&mut encoder).unwrap(); + let bytes = encoder.finish(); + assert!(bytes.len() < 256 * 1024); + let mut decoder = BoundedDecoder::new(&bytes, INPUT_LIMIT).unwrap(); + let decoded = ObjectBatch::decode(&mut decoder).unwrap(); + decoder.finish().unwrap(); + let mut encoder = BoundedEncoder::new(INPUT_LIMIT).unwrap(); + decoded.encode(&mut encoder).unwrap(); + assert_eq!(encoder.finish(), bytes); +} diff --git a/src/object_reads.rs b/src/object_reads.rs index d455b47..e8d6dee 100644 --- a/src/object_reads.rs +++ b/src/object_reads.rs @@ -253,6 +253,17 @@ fn decode_object(row: Vec) -> cellule_runtime::Result { { ObjectStorage::Inline(body) } + ("packed", SqlValue::Null, SqlValue::Blob(pack), SqlValue::Null) + if kind == ObjectKind::Blob && size >= 0 => + { + ObjectStorage::Packed { + size: size as u64, + blake3: digest, + pack: pack + .try_into() + .map_err(|_| Error::Command("invalid pack locator"))?, + } + } ("external", SqlValue::Null, SqlValue::Blob(sha256), SqlValue::Null) if kind == ObjectKind::Blob && size >= 0 => { diff --git a/src/pack_store.rs b/src/pack_store.rs new file mode 100644 index 0000000..708b011 --- /dev/null +++ b/src/pack_store.rs @@ -0,0 +1,439 @@ +//! Immutable pack artifacts; SQLite retains canonical identities and closure. +use crate::{ + ObjectId, ObjectKind, RepositoryCell, + git_cache::GitCache, + git_gateway::GatewayError, + git_http::GitProcess, + large_blob::{LargeBlobReference, LargeBlobStore}, +}; +use cellule_ltx::DiskBudget; +use cellule_runtime::{ + Error, InvocationError, MutationIdentity, + primitives::sql::{SqlBatch, SqlResultSet, SqlStatement, SqlValue}, +}; +use object_store::ObjectStore; +use std::{collections::BTreeMap, path::PathBuf, sync::Arc}; +use tokio::sync::Mutex; + +#[derive(Clone)] +pub(crate) struct PackRecord { + pub hash: ObjectId, + pub pack: LargeBlobReference, + pub index: LargeBlobReference, + pub approved: bool, + pub covered_through: i64, +} + +pub(crate) struct PackReader { + store: LargeBlobStore, + root: PathBuf, + budget: DiskBudget, + format: crate::ObjectFormat, + cache: Mutex>>, + installation: Mutex<()>, + private: Mutex)>>, +} +impl PackReader { + pub(crate) fn new( + store: Arc, + repository: [u8; 16], + root: PathBuf, + budget: DiskBudget, + format: crate::ObjectFormat, + ) -> Self { + Self { + store: LargeBlobStore::new(store, repository), + root, + budget, + format, + cache: Mutex::new(None), + installation: Mutex::new(()), + private: Mutex::new(None), + } + } + pub(crate) async fn cache(&self) -> Result, GatewayError> { + let mut cache = self.cache.lock().await; + if cache.is_none() { + *cache = Some( + GitCache::create( + self.root.clone(), + self.budget.clone(), + "refs/heads/main", + self.format, + ) + .await?, + ); + } + Ok(Arc::clone( + cache.as_ref().ok_or(GatewayError::MalformedCache)?, + )) + } + pub(crate) async fn replace(&self, old: &Arc, next: Arc) { + let mut cache = self.cache.lock().await; + if cache.as_ref().is_some_and(|cache| Arc::ptr_eq(cache, old)) { + *cache = Some(next); + } + } + pub(crate) async fn install( + &self, + cache: &Arc, + record: &PackRecord, + ) -> Result<(), GatewayError> { + let _installation = self.installation.lock().await; + if cache.has_durable_pack(record.pack.sha256) { + return Ok(()); + } + cache + .install_pack( + record.hash, + self.store.read(&record.pack).await?, + self.store.read(&record.index).await?, + ) + .await?; + if record.approved { + cache.mark_durable_pack(record.pack.sha256); + } + Ok(()) + } + pub(crate) async fn native_reader( + &self, + record: PackRecord, + oid: ObjectId, + size: u64, + digest: [u8; 32], + ) -> Result { + // Incomplete archives stay in one bounded private cache. Their foreign + // members cannot enter a transport cache; each extracted body is checked. + let cache = if record.approved { + self.cache().await? + } else { + let mut private = self.private.lock().await; + if private + .as_ref() + .is_none_or(|(sha, _)| *sha != record.pack.sha256) + { + *private = Some(( + record.pack.sha256, + GitCache::create( + self.root.clone(), + self.budget.clone(), + "refs/heads/main", + self.format, + ) + .await?, + )); + } + Arc::clone(&private.as_ref().ok_or(GatewayError::MalformedCache)?.1) + }; + self.install(&cache, &record).await?; + if !record.approved { + cache.mark_durable_pack(record.pack.sha256); + } + let mut command = crate::native_git::command(&cache.git_dir())?; + command + .args(["cat-file", "blob", &hex::encode(oid)]) + .stdout(std::process::Stdio::piped()) + .stderr(std::process::Stdio::null()); + let mut process = GitProcess::spawn(command, Arc::clone(&cache))?; + let output = process + .child + .stdout + .take() + .ok_or(GatewayError::MalformedCache)?; + Ok(NativePackedRead { + process, + output, + oid, + size, + remaining: size, + expected: digest, + canonical: Some(crate::git_format::ObjectHasher::new( + oid.format(), + ObjectKind::Blob, + size, + )), + digest: blake3::Hasher::new(), + finished: false, + }) + } + pub(crate) async fn read_blob( + &self, + record: PackRecord, + oid: ObjectId, + size: u64, + digest: [u8; 32], + ) -> Result, GatewayError> { + if size > crate::INLINE_OBJECT_LIMIT as u64 { + return Err(GatewayError::MalformedCache); + } + let mut reader = self.native_reader(record, oid, size, digest).await?; + let mut body = Vec::with_capacity(size as usize); + while let Some(bytes) = reader.next().await? { + body.extend_from_slice(&bytes); + } + Ok(body) + } + pub(crate) async fn upload(&self, path: PathBuf) -> Result { + let hash_path = path.clone(); + let (oid, size) = tokio::task::spawn_blocking(move || { + use std::io::Read; + let mut file = std::fs::File::open(hash_path)?; + let size = file.metadata()?.len(); + let mut hash = crate::git_format::ObjectHasher::new( + crate::ObjectFormat::Sha256, + ObjectKind::Blob, + size, + ); + let mut buffer = vec![0; 8 << 20]; + loop { + let count = file.read(&mut buffer)?; + if count == 0 { + break; + } + hash.update(&buffer[..count]); + } + Ok::<_, std::io::Error>((hash.finalize(), size)) + }) + .await??; + Ok(self + .store + .put(oid, size, &mut tokio::fs::File::open(path).await?) + .await?) + } +} + +/// Constant-memory extraction used only when an incomplete pack cannot be +/// admitted as a whole. Verification precedes the final returned body range. +pub(crate) struct NativePackedRead { + process: GitProcess>, + output: tokio::process::ChildStdout, + pub(crate) oid: ObjectId, + pub(crate) size: u64, + remaining: u64, + expected: [u8; 32], + canonical: Option, + digest: blake3::Hasher, + finished: bool, +} +impl NativePackedRead { + pub(crate) async fn next(&mut self) -> Result, GatewayError> { + use tokio::io::AsyncReadExt; + if self.finished { + return Ok(None); + } + let mut bytes = vec![0; self.remaining.min(8 << 20) as usize]; + tokio::time::timeout( + std::time::Duration::from_secs(120), + self.output.read_exact(&mut bytes), + ) + .await + .map_err(|_| crate::git_http::GitHttpError::Timeout)??; + self.remaining -= bytes.len() as u64; + self.canonical + .as_mut() + .ok_or(GatewayError::MalformedCache)? + .update(&bytes); + self.digest.update(&bytes); + if self.remaining == 0 { + let mut trailing = [0]; + let eof = tokio::time::timeout( + std::time::Duration::from_secs(120), + self.output.read(&mut trailing), + ) + .await + .map_err(|_| crate::git_http::GitHttpError::Timeout)??; + if eof != 0 + || self + .canonical + .take() + .ok_or(GatewayError::MalformedCache)? + .finalize() + != self.oid + || self.digest.finalize().as_bytes() != &self.expected + { + return Err(GatewayError::MalformedCache); + } + let status = tokio::time::timeout( + std::time::Duration::from_secs(120), + self.process.child.wait(), + ) + .await + .map_err(|_| crate::git_http::GitHttpError::Timeout)??; + self.process.disarm(); + if !status.success() { + return Err(GatewayError::MalformedCache); + } + self.finished = true; + } + Ok(Some(bytes.into())) + } +} + +const COLUMNS: &str = "sha256, pack_hash, pack_oid, pack_size, pack_digest, index_oid, index_size, index_digest, index_sha256, approved, covered_through"; +type ReadError = InvocationError>; +fn invalid() -> ReadError { + InvocationError::NotStarted(Error::Command("invalid durable pack record")) +} +fn decode(row: &[SqlValue]) -> Result { + let [ + SqlValue::Blob(sha), + SqlValue::Blob(hash), + SqlValue::Blob(pack_oid), + SqlValue::Integer(pack_size), + SqlValue::Blob(pack_digest), + SqlValue::Blob(index_oid), + SqlValue::Integer(index_size), + SqlValue::Blob(index_digest), + SqlValue::Blob(index_sha), + SqlValue::Integer(approved), + SqlValue::Integer(covered_through), + ] = row + else { + return Err(invalid()); + }; + Ok(PackRecord { + hash: hash.as_slice().try_into().map_err(|_| invalid())?, + pack: LargeBlobReference { + oid: pack_oid.as_slice().try_into().map_err(|_| invalid())?, + size: (*pack_size).try_into().map_err(|_| invalid())?, + blake3: pack_digest.as_slice().try_into().map_err(|_| invalid())?, + sha256: sha.as_slice().try_into().map_err(|_| invalid())?, + }, + index: LargeBlobReference { + oid: index_oid.as_slice().try_into().map_err(|_| invalid())?, + size: (*index_size).try_into().map_err(|_| invalid())?, + blake3: index_digest.as_slice().try_into().map_err(|_| invalid())?, + sha256: index_sha.as_slice().try_into().map_err(|_| invalid())?, + }, + approved: *approved == 1, + covered_through: *covered_through, + }) +} +impl RepositoryCell { + pub(crate) async fn pack_record(&self, sha: [u8; 32]) -> Result { + let result = self + .sql + .query( + None, + SqlBatch { + statements: vec![SqlStatement { + sql: format!("SELECT {COLUMNS} FROM git_packs WHERE sha256 = ?1"), + parameters: vec![SqlValue::Blob(sha.to_vec())], + }], + }, + ) + .await?; + decode( + result + .output + .first() + .and_then(|set| set.rows.first()) + .ok_or_else(invalid)?, + ) + } + pub(crate) async fn approved_packs(&self, after: &[u8]) -> Result, ReadError> { + let result = self.sql.query(None, SqlBatch { statements: vec![SqlStatement { sql: format!("SELECT {COLUMNS} FROM git_packs WHERE approved = 1 AND sha256 > ?1 ORDER BY sha256 LIMIT 128"), parameters: vec![SqlValue::Blob(after.to_vec())] }] }).await?; + result + .output + .first() + .ok_or_else(invalid)? + .rows + .iter() + .map(|row| decode(row)) + .collect() + } + pub(crate) async fn register_pack( + &self, + identity: MutationIdentity, + record: &PackRecord, + ) -> Result<(), ReadError> { + let values = vec![ + SqlValue::Blob(record.pack.sha256.to_vec()), + SqlValue::Blob(record.hash.to_vec()), + SqlValue::Blob(record.pack.oid.to_vec()), + SqlValue::Integer(record.pack.size.try_into().map_err(|_| invalid())?), + SqlValue::Blob(record.pack.blake3.to_vec()), + SqlValue::Blob(record.index.oid.to_vec()), + SqlValue::Integer(record.index.size.try_into().map_err(|_| invalid())?), + SqlValue::Blob(record.index.blake3.to_vec()), + SqlValue::Blob(record.index.sha256.to_vec()), + ]; + let result = self.sql.batch(identity, SqlBatch { statements: vec![ + SqlStatement { sql: "INSERT INTO git_packs (sha256, pack_hash, pack_oid, pack_size, pack_digest, index_oid, index_size, index_digest, index_sha256) VALUES (?1,?2,?3,?4,?5,?6,?7,?8,?9) ON CONFLICT DO NOTHING".into(), parameters: values }, + ] }).await?; + drop(result); + let stored = self.pack_record(record.pack.sha256).await?; + if stored.hash != record.hash + || stored.pack.oid != record.pack.oid + || stored.pack.size != record.pack.size + || stored.pack.blake3 != record.pack.blake3 + || stored.index.oid != record.index.oid + || stored.index.size != record.index.size + || stored.index.blake3 != record.index.blake3 + || stored.index.sha256 != record.index.sha256 + { + return Err(invalid()); + } + Ok(()) + } + pub(crate) async fn approve_pack( + &self, + identity: MutationIdentity, + sha: [u8; 32], + verified_count: usize, + ) -> Result<(), ReadError> { + self.sql + .batch( + identity, + SqlBatch { + statements: vec![SqlStatement { + // Every unique index member has a matching canonical SQL row. + // Equal cardinality proves this pack covers the entire immutable + // object table at this transaction, without scanning it on recovery. + sql: "UPDATE git_packs SET approved = 1, covered_through = CASE WHEN ?2 = (SELECT COUNT(oid) FROM objects) THEN (SELECT COALESCE(MAX(sequence), 0) FROM objects) ELSE covered_through END WHERE sha256 = ?1".into(), + parameters: vec![SqlValue::Blob(sha.to_vec()), SqlValue::Integer(verified_count.try_into().map_err(|_| invalid())?)], + }], + }, + ) + .await?; + Ok(()) + } + pub(crate) async fn canonical_headers( + &self, + ids: &[ObjectId], + ) -> Result, ReadError> { + if ids.is_empty() || ids.len() > crate::object_batch::MAX_BATCH_OBJECTS { + return Err(invalid()); + } + let placeholders = vec!["?"; ids.len()].join(","); + let result = self.sql.query(None, SqlBatch { statements: vec![SqlStatement { sql: format!("SELECT oid, kind, size, digest FROM objects WHERE oid IN ({placeholders})"), parameters: ids.iter().map(|oid| SqlValue::Blob(oid.to_vec())).collect() }] }).await?; + let mut headers = BTreeMap::new(); + for row in &result.output.first().ok_or_else(invalid)?.rows { + let [ + SqlValue::Blob(oid), + SqlValue::Text(kind), + SqlValue::Integer(size), + SqlValue::Blob(digest), + ] = row.as_slice() + else { + return Err(invalid()); + }; + let kind = match kind.as_str() { + "blob" => ObjectKind::Blob, + "tree" => ObjectKind::Tree, + "commit" => ObjectKind::Commit, + "tag" => ObjectKind::Tag, + _ => return Err(invalid()), + }; + headers.insert( + oid.as_slice().try_into().map_err(|_| invalid())?, + ( + kind, + (*size).try_into().map_err(|_| invalid())?, + digest.as_slice().try_into().map_err(|_| invalid())?, + ), + ); + } + Ok(headers) + } +} diff --git a/src/schema.sql b/src/schema.sql index 7462f5d..0948f0c 100644 --- a/src/schema.sql +++ b/src/schema.sql @@ -15,7 +15,7 @@ CREATE TABLE objects ( kind TEXT NOT NULL CHECK(kind IN ('blob', 'tree', 'commit', 'tag')), size INTEGER NOT NULL CHECK(size >= 0), digest BLOB NOT NULL CHECK(length(digest) = 32), - storage TEXT NOT NULL CHECK(storage IN ('inline', 'external', 'chunked')), + storage TEXT NOT NULL CHECK(storage IN ('inline', 'external', 'chunked', 'packed')), body BLOB, external_sha256 BLOB, chunk_id BLOB UNIQUE REFERENCES object_uploads(id) CHECK(chunk_id IS NULL OR length(chunk_id) = 16), @@ -24,6 +24,8 @@ CREATE TABLE objects ( OR (storage = 'external' AND kind = 'blob' AND body IS NULL AND chunk_id IS NULL AND length(external_sha256) = 32) OR + (storage = 'packed' AND kind = 'blob' AND body IS NULL AND chunk_id IS NULL AND length(external_sha256) = 32 ) + OR (storage = 'chunked' AND kind != 'blob' AND body IS NULL AND external_sha256 IS NULL AND chunk_id IS NOT NULL AND size > 786432) ) ); @@ -321,3 +323,14 @@ CREATE TABLE pull_thread_comments ( created_ms INTEGER NOT NULL CHECK(created_ms >= 0) ); CREATE INDEX comments_by_thread ON pull_thread_comments(thread_number, number); + +-- Immutable uploaded pack/index pairs. Approval follows canonical verification +-- of every native index member, including pre-existing objects and thin bases. +CREATE TABLE git_packs ( + sha256 BLOB PRIMARY KEY CHECK(length(sha256) = 32), + pack_hash BLOB NOT NULL UNIQUE CHECK(length(pack_hash) IN (20, 32)), + pack_oid BLOB NOT NULL, pack_size INTEGER NOT NULL CHECK(pack_size > 0), pack_digest BLOB NOT NULL CHECK(length(pack_digest) = 32), + index_oid BLOB NOT NULL, index_size INTEGER NOT NULL CHECK(index_size > 0), index_digest BLOB NOT NULL CHECK(length(index_digest) = 32), index_sha256 BLOB NOT NULL CHECK(length(index_sha256) = 32), + approved INTEGER NOT NULL DEFAULT 0 CHECK(approved IN (0, 1)), + covered_through INTEGER NOT NULL DEFAULT 0 CHECK(covered_through >= 0) +) WITHOUT ROWID; diff --git a/src/server.rs b/src/server.rs index 834f186..aaf4ce3 100644 --- a/src/server.rs +++ b/src/server.rs @@ -161,6 +161,7 @@ struct RunningServer { stop: CancellationToken, renewal: AbortOnDropHandle>, ingress_stop: CancellationToken, + maintenance_stop: CancellationToken, release_stop: CancellationToken, serving: JoinHandle>, ssh_serving: Option>>, @@ -190,6 +191,7 @@ pub(crate) struct RepositoryManager { residency_admission: AccountAdmission, transfers: AccountAdmission, tasks: TaskTracker, + maintenance_stop: CancellationToken, } pub(crate) enum MembershipOutcome { @@ -494,6 +496,7 @@ impl RunningServer { .build()?, ); let stop = CancellationToken::new(); + let maintenance_stop = CancellationToken::new(); node.install_task_group(CancellationToken::new(), release_stop.clone())?; // Preflight may outlast a lease. Start its lifetime only when enrollment // begins, so slow probing cannot publish an already-expired advertisement. @@ -599,6 +602,7 @@ impl RunningServer { "account repository activations", ), tasks: tasks.clone(), + maintenance_stop: maintenance_stop.clone(), }); let api = Arc::new(RepositoryHttp::new(Arc::clone(&manager), tasks.clone())); deployment.require_ready().await?; @@ -661,6 +665,7 @@ impl RunningServer { stop, renewal, ingress_stop, + maintenance_stop, release_stop, serving, tasks, diff --git a/src/server/lifecycle.rs b/src/server/lifecycle.rs index f3f985d..c613102 100644 --- a/src/server/lifecycle.rs +++ b/src/server/lifecycle.rs @@ -89,6 +89,7 @@ impl CanopyServer { impl RunningServer { async fn shutdown(self) -> Result<(), ServerError> { + self.maintenance_stop.cancel(); self.ingress_stop.cancel(); let serving = self.serving.await; let ssh_serving = if let Some(task) = self.ssh_serving { diff --git a/src/server/residency.rs b/src/server/residency.rs index 8a5f61d..71a6a04 100644 --- a/src/server/residency.rs +++ b/src/server/residency.rs @@ -497,12 +497,34 @@ impl RepositoryManager { .with_signer_directory(Arc::clone(&self.directory)), ); let router = self.router_for(entry, Arc::clone(&gateway))?; + let pin = Arc::new(()); + let weak_gateway = Arc::downgrade(&gateway); + let weak_pin = Arc::downgrade(&pin); + let stop = self.maintenance_stop.clone(); + self.tasks.spawn(async move { + let mut interval = tokio::time::interval(std::time::Duration::from_secs(60)); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + interval.tick().await; + loop { + tokio::select! { + () = stop.cancelled() => return, + _ = interval.tick() => {}, + } + let (Some(gateway), Some(_pin)) = (weak_gateway.upgrade(), weak_pin.upgrade()) else { return; }; + tokio::select! { + () = stop.cancelled() => return, + result = gateway.maintain() => { + if let Err(error) = result { tracing::warn!(error = ?error, "background Git maintenance failed; previous cache retained"); } + } + } + } + }); Ok(LoadedRepository { repository, gateway, name: entry.name.clone(), router, - pin: Arc::new(()), + pin, last_used: Instant::now(), initialized: false, local, diff --git a/tests/multi_server/backup.rs b/tests/multi_server/backup.rs index 73eff68..506a8c9 100644 --- a/tests/multi_server/backup.rs +++ b/tests/multi_server/backup.rs @@ -53,6 +53,13 @@ async fn backup_restores_git_lfs_and_collaboration_without_original_storage() -> let lfs = vec![29_u8; 256 * 1024]; std::fs::write(local.join("large.bin"), &blob)?; std::fs::write(local.join("data.lfs"), &lfs)?; + // Force a native receive pack with delta-compressed small blobs. Backup + // must retain both immutable artifacts after original storage is deleted. + for n in 0..200 { + let mut body = vec![b'x'; 32 * 1024]; + body[..8].copy_from_slice(&(n as u64).to_le_bytes()); + std::fs::write(local.join(format!("packed-{n:03}")), body)?; + } run_git(Some(&local), &["add", "."]).await?; run_git(Some(&local), &["commit", "-m", "Back up all bytes"]).await?; run_git( @@ -90,7 +97,7 @@ async fn backup_restores_git_lfs_and_collaboration_without_original_storage() -> let report = deployment .create_backup(id, backup.clone(), worker()) .await?; - assert_eq!(report.external_objects, 2); + assert_eq!(report.external_objects, 3); assert_eq!(report.cells, 2); deployment .create_backup(id, backup.clone(), worker()) @@ -172,6 +179,10 @@ async fn backup_restores_git_lfs_and_collaboration_without_original_storage() -> std::fs::read(cloned.join("data.lfs"))? == lfs, "restored LFS body differs" ); + assert_eq!( + std::fs::read(cloned.join("packed-199"))?, + std::fs::read(local.join("packed-199"))? + ); run_git(Some(&cloned), &["fsck", "--strict", "--full"]).await?; let restored_issue: Value = client .get(format!( diff --git a/tests/multi_server/large_objects.rs b/tests/multi_server/large_objects.rs index b027455..8e9e789 100644 --- a/tests/multi_server/large_objects.rs +++ b/tests/multi_server/large_objects.rs @@ -3,6 +3,9 @@ use super::*; #[tokio::test(flavor = "multi_thread")] async fn large_tree_commit_and_tag_restore_from_sqlite_after_owner_restart() -> Result<(), Box> { + let _ = tracing_subscriber::fmt() + .with_env_filter(tracing_subscriber::EnvFilter::from_default_env()) + .try_init(); let store: Arc = Arc::new(InMemory::new()); let workspace = tempfile::TempDir::new()?; let local = workspace.path().join("source"); diff --git a/tests/repository_cell/graph.rs b/tests/repository_cell/graph.rs index a8e1ce3..8d2a9c4 100644 --- a/tests/repository_cell/graph.rs +++ b/tests/repository_cell/graph.rs @@ -126,10 +126,15 @@ pub async fn verify( Err(InvocationError::Rejected(_)) )); assert!( - !certificates(sql) + certificates(sql) .await? .contains(&SqlValue::Blob(blob.to_vec())) ); + assert!( + !certificates(sql) + .await? + .contains(&SqlValue::Blob(tree.to_vec())) + ); } // A rejected batch must roll back edges as well as closure certificates; // otherwise a staged tree could authorize a later raw-OID fetch. @@ -327,7 +332,7 @@ async fn resumable(repository: &RepositoryCell, sql: &SqlCell) .iter() .filter(|(oid, _)| persisted.contains(&SqlValue::Blob(oid.to_vec()))) .count(), - 256 + leaves.len() ); assert!(!persisted.contains(&SqlValue::Blob(root.to_vec()))); assert_eq!( diff --git a/tests/repository_cell/pages.rs b/tests/repository_cell/pages.rs index 00f21c8..2b38b28 100644 --- a/tests/repository_cell/pages.rs +++ b/tests/repository_cell/pages.rs @@ -43,6 +43,7 @@ async fn clear(sql: &SqlCell) -> Result { identity()?, SqlBatch { statements: [ + "DELETE FROM object_closure", "DELETE FROM objects", "DELETE FROM object_chunks", "DELETE FROM object_uploads", From dd76191261fb90df1b2b484dc0c368fe0c4e2b82 Mon Sep 17 00:00:00 2001 From: forhappy Date: Thu, 1 Oct 2026 00:43:09 -0700 Subject: [PATCH 2/6] Bound graph edge statements within larger certificate batches --- src/graph.rs | 4 ++-- tests/multi_server/backup.rs | 3 +++ 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/src/graph.rs b/src/graph.rs index 1ab5c20..652f41c 100644 --- a/src/graph.rs +++ b/src/graph.rs @@ -144,7 +144,7 @@ impl Command for CertifyObjects { // Only certified, typed edges enter the reachability index. Keeping // this with closure publication prevents lazy fetch from trusting // staged or malformed object graphs. - for edges in edges.chunks(MAX_CERTIFICATES) { + for edges in edges.chunks(crate::object_batch::MAX_OBJECTS) { context.sql(&SqlBatch { statements: edges .iter() @@ -164,7 +164,7 @@ impl Command for CertifyObjects { .filter(|(_, kind)| *kind == Some(ObjectKind::Commit)) .map(|(parent, _)| *parent) .collect(); - for parents in parents.chunks(MAX_CERTIFICATES) { + for parents in parents.chunks(crate::object_batch::MAX_OBJECTS) { context.sql(&SqlBatch { statements: parents.iter().map(|parent| SqlStatement { sql: "INSERT INTO commit_parents (child, parent) VALUES (?1, ?2) ON CONFLICT DO NOTHING".into(), parameters: vec![SqlValue::Blob(oid.to_vec()), SqlValue::Blob(parent.to_vec())], diff --git a/tests/multi_server/backup.rs b/tests/multi_server/backup.rs index 506a8c9..11bea0c 100644 --- a/tests/multi_server/backup.rs +++ b/tests/multi_server/backup.rs @@ -13,6 +13,9 @@ type Result = std::result::Result>; #[tokio::test(flavor = "multi_thread")] async fn backup_restores_git_lfs_and_collaboration_without_original_storage() -> Result { + let _ = tracing_subscriber::fmt() + .with_env_filter(tracing_subscriber::EnvFilter::from_default_env()) + .try_init(); let store: Arc = Arc::new(InMemory::new()); let files = tempfile::TempDir::new()?; let address = available_address().await?; From 223b88879d2ba01707c5345924b419a9ebd5583e Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 2 Oct 2026 15:20:06 -0700 Subject: [PATCH 3/6] Adapt pack qualification and lints to the Rust workspace --- .../src/git_cache/maintenance.rs | 2 +- crates/canopy-server/src/git_gateway/fetch.rs | 20 +++++++------- crates/canopy-server/src/git_gateway/mod.rs | 8 +++--- crates/canopy-server/src/pack_store.rs | 2 +- scripts/benchmark_large_repository.py | 17 +++++++++--- scripts/test_benchmark_large_repository.py | 27 +++++++++++++++++++ 6 files changed, 55 insertions(+), 21 deletions(-) create mode 100644 scripts/test_benchmark_large_repository.py diff --git a/crates/canopy-server/src/git_cache/maintenance.rs b/crates/canopy-server/src/git_cache/maintenance.rs index c4cc823..d8da15e 100644 --- a/crates/canopy-server/src/git_cache/maintenance.rs +++ b/crates/canopy-server/src/git_cache/maintenance.rs @@ -500,7 +500,7 @@ impl GitCache { Ok::<_, io::Error>(()) }) .await - .map_err(|error| io::Error::other(error))??; + .map_err(io::Error::other)??; Ok::<_, GitHttpError>(()) }; let result = tokio::time::timeout(WORKER_DEADLINE, run) diff --git a/crates/canopy-server/src/git_gateway/fetch.rs b/crates/canopy-server/src/git_gateway/fetch.rs index fa1a060..27184dd 100644 --- a/crates/canopy-server/src/git_gateway/fetch.rs +++ b/crates/canopy-server/src/git_gateway/fetch.rs @@ -210,17 +210,15 @@ impl GitGateway { .get(1) .and_then(|set| set.rows.first()) .map(Vec::as_slice), - ) { - if usize::try_from(*count).ok() == Some(shared.packed_count()) { - // Never invert build_cache's lock order by waiting here. - if let Ok(mut objects) = self.objects.try_lock() { - if let Some(objects) = objects - .as_mut() - .filter(|objects| Arc::ptr_eq(&objects.cache, &shared)) - { - objects.through = objects.through.max(*high_water); - } - } + ) && usize::try_from(*count).ok() == Some(shared.packed_count()) + { + // Never invert build_cache's lock order by waiting here. + if let Ok(mut objects) = self.objects.try_lock() + && let Some(objects) = objects + .as_mut() + .filter(|objects| Arc::ptr_eq(&objects.cache, &shared)) + { + objects.through = objects.through.max(*high_water); } } } diff --git a/crates/canopy-server/src/git_gateway/mod.rs b/crates/canopy-server/src/git_gateway/mod.rs index b22e415..98407da 100644 --- a/crates/canopy-server/src/git_gateway/mod.rs +++ b/crates/canopy-server/src/git_gateway/mod.rs @@ -627,10 +627,10 @@ impl GitGateway { ); } Ok(retained) => { - if retained == count { - if let Some(record) = &archive { - shared.cache.mark_durable_pack(record.pack.sha256); - } + if retained == count + && let Some(record) = &archive + { + shared.cache.mark_durable_pack(record.pack.sha256); } } Err(error) => { diff --git a/crates/canopy-server/src/pack_store.rs b/crates/canopy-server/src/pack_store.rs index 71df3a9..042929f 100644 --- a/crates/canopy-server/src/pack_store.rs +++ b/crates/canopy-server/src/pack_store.rs @@ -1,10 +1,10 @@ //! Immutable pack artifacts; SQLite retains canonical identities and closure. use crate::{ ObjectId, ObjectKind, RepositoryCell, + blob::{LargeBlobReference, LargeBlobStore}, git_cache::GitCache, git_gateway::GatewayError, git_http::GitProcess, - blob::{LargeBlobReference, LargeBlobStore}, }; use cellule_ltx::DiskBudget; use cellule_runtime::{ diff --git a/scripts/benchmark_large_repository.py b/scripts/benchmark_large_repository.py index 90e1b46..43f8279 100644 --- a/scripts/benchmark_large_repository.py +++ b/scripts/benchmark_large_repository.py @@ -24,6 +24,18 @@ import local_eval +def source_tree_digest(root): + digest = hashlib.sha256() + sources = [root / "Cargo.toml", root / "Cargo.lock"] + sources.extend(path for path in (root / "crates").rglob("*") + if path.is_file() and (path.suffix in (".rs", ".sql") + or path.name == "Cargo.toml")) + for source_file in sorted(sources): + digest.update(str(source_file.relative_to(root)).encode() + b"\0" + + source_file.read_bytes()) + return digest.hexdigest() + + def references(directory): output = local_eval.run("git", "-C", str(directory), "for-each-ref", "--format=%(objectname) %(refname)", "refs/heads", "refs/tags") @@ -72,14 +84,11 @@ def main(): GIT_CONFIG_VALUE_0="", GIT_CONFIG_KEY_1="http.extraHeader", GIT_CONFIG_VALUE_1="Authorization: Basic " + auth, GIT_TERMINAL_PROMPT="0", GIT_LFS_SKIP_SMUDGE="1") - source_digest = hashlib.sha256() - for source_file in sorted([*Path("src").rglob("*.rs"), *Path("src").rglob("*.sql"), Path("Cargo.toml"), Path("Cargo.lock")]): - source_digest.update(str(source_file).encode() + b"\0" + source_file.read_bytes()) report = { "mode": args.mode, "name": args.name, "status": "running", "stages": [], "source": str(args.source.resolve()), "canopy_binary_sha256": metadata["binary_sha256"], "canopy_revision": local_eval.run("git", "rev-parse", "HEAD"), - "canopy_source_tree_sha256": source_digest.hexdigest(), + "canopy_source_tree_sha256": source_tree_digest(Path(__file__).resolve().parent.parent), "canopy_working_tree_dirty": bool(local_eval.run("git", "status", "--porcelain")), "provider_image": metadata["provider_image"], "git_version": local_eval.run("git", "--version"), "host_platform": platform.platform(), "host_cpu_count": os.cpu_count(), diff --git a/scripts/test_benchmark_large_repository.py b/scripts/test_benchmark_large_repository.py new file mode 100644 index 0000000..8e6b12a --- /dev/null +++ b/scripts/test_benchmark_large_repository.py @@ -0,0 +1,27 @@ +"""Qualification evidence must identify code in every workspace crate.""" +from pathlib import Path +import tempfile +import unittest + +from benchmark_large_repository import source_tree_digest + + +class SourceIdentityTests(unittest.TestCase): + def test_workspace_code_and_manifest_changes_invalidate_identity(self): + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + (root / "Cargo.toml").write_text("[workspace]") + (root / "Cargo.lock").write_text("lock") + for name in ("canopy-server", "canopy-git-format", "canopy-object-storage"): + crate = root / "crates" / name + (crate / "src").mkdir(parents=True) + for path in (crate / "Cargo.toml", crate / "src" / "lib.rs", + crate / "src" / "schema.sql"): + path.write_text("before") + before = source_tree_digest(root) + path.write_text("after") + self.assertNotEqual(before, source_tree_digest(root), str(path)) + + +if __name__ == "__main__": + unittest.main() From cf3dab1c01f326350af7d739dbfe148f15b0170d Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 2 Oct 2026 15:28:09 -0700 Subject: [PATCH 4/6] Keep repository upgrade rejection separate from directory compatibility --- .../tests/directory_cell/compatibility.rs | 35 ++++++++++++++++++- docs/kubernetes-qualification.md | 6 ++++ 2 files changed, 40 insertions(+), 1 deletion(-) diff --git a/crates/canopy-server/tests/directory_cell/compatibility.rs b/crates/canopy-server/tests/directory_cell/compatibility.rs index c8d37db..7ed6ccb 100644 --- a/crates/canopy-server/tests/directory_cell/compatibility.rs +++ b/crates/canopy-server/tests/directory_cell/compatibility.rs @@ -17,7 +17,8 @@ fn bounded_authentication_retains_the_exact_selected_predecessor() "directory-compatibility-test", ))?; let registry = application.registry(); - registry.verify_rolling_from(previous)?; + // Directory compatibility remains independent of the repository pack + // schema. Whole-release rolling compatibility is fenced separately below. let predecessor: serde_json::Value = serde_json::from_slice(previous)?; let old_directory = predecessor["modules"] .as_array() @@ -59,3 +60,35 @@ fn bounded_authentication_retains_the_exact_selected_predecessor() assert_eq!(queries[2]["output_limit"], 256); Ok(()) } + +#[test] +fn packed_repository_rejects_an_unqualified_predecessor_upgrade() +-> Result<(), Box> { + let previous = include_bytes!("fixtures/c51-selected-release.json").trim_ascii_end(); + let application = CanopyApplication::compile(build_descriptor( + include_bytes!("../../../../Cargo.lock"), + "repository-upgrade-fence-test", + ))?; + let registry = application.registry(); + let predecessor: serde_json::Value = serde_json::from_slice(previous)?; + let old_repository = predecessor["modules"] + .as_array() + .ok_or("modules missing")? + .iter() + .find(|module| module["name"] == "repository") + .ok_or("repository missing")?; + let bytes: [u8; 32] = hex::decode(old_repository["code"].as_str().ok_or("code missing")?)? + .try_into() + .map_err(|_| "invalid predecessor code length")?; + assert!(!registry.supports_cell( + canopy_server::REPOSITORIES, + CatalogRole::Sql, + Digest::from_bytes(bytes), + 1, + )); + assert!( + registry.verify_rolling_from(previous).is_err(), + "pack storage requires a qualified repository migration before rolling upgrade" + ); + Ok(()) +} diff --git a/docs/kubernetes-qualification.md b/docs/kubernetes-qualification.md index c668bc4..f121b4e 100644 --- a/docs/kubernetes-qualification.md +++ b/docs/kubernetes-qualification.md @@ -53,6 +53,12 @@ Durable pack compaction/garbage collection, preview schema migration, provider fault tests and bounded Linux concurrent-load qualification remain production work. See [the operational recipe](../deploy/local-evaluation.md). +The directory retains the selected c51 release's credential code, but the +repository pack schema does not support a rolling upgrade from that release. +Compatibility tests assert that its repository code and whole-release upgrade +are rejected. Existing deployments need a qualified migration; local evaluation +uses a fresh provider prefix. + Current evidence: ```text From 894a678a4447db515580106a6a7be264d2bc1e9f Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 2 Oct 2026 15:57:30 -0700 Subject: [PATCH 5/6] Align retained-directory and graph fixtures with pack storage contracts --- crates/canopy-server/tests/multi_server/retained_catalog.rs | 4 +++- crates/canopy-server/tests/repository_cell/graph.rs | 2 +- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/crates/canopy-server/tests/multi_server/retained_catalog.rs b/crates/canopy-server/tests/multi_server/retained_catalog.rs index d296fbd..7bab047 100644 --- a/crates/canopy-server/tests/multi_server/retained_catalog.rs +++ b/crates/canopy-server/tests/multi_server/retained_catalog.rs @@ -79,7 +79,8 @@ async fn retained_fixture() -> Result { blake3::hash(previous).to_hex().as_str(), "e31bf1a951e2fa19d91e9f964b2ddeade1a81b05a20ad628362819a1487c16b1" ); - registry.verify_rolling_from(previous)?; + // This explicit fixture activation carries only a Directory Cell. The + // repository pack schema separately rejects whole-release rolling upgrade. let descriptor: serde_json::Value = serde_json::from_slice(previous)?; let module = descriptor["modules"] .as_array() @@ -92,6 +93,7 @@ async fn retained_fixture() -> Result { .try_into() .map_err(|_| "invalid predecessor code")?, ); + assert!(registry.supports_cell(directory::DIRECTORY, CatalogRole::Sql, old_code, 1)); let releases = ReleaseStore::new(layout.clone(), identity)?; let image = format!("sha256:{}", hex::encode(configuration.image.as_bytes())); let old_operation = RequestId::from_bytes(uuid::Uuid::new_v4().into_bytes()); diff --git a/crates/canopy-server/tests/repository_cell/graph.rs b/crates/canopy-server/tests/repository_cell/graph.rs index 8d2a9c4..abe58a5 100644 --- a/crates/canopy-server/tests/repository_cell/graph.rs +++ b/crates/canopy-server/tests/repository_cell/graph.rs @@ -394,7 +394,7 @@ struct CertificateCommand; impl cellule_runtime::Command for CertificateCommand { const MODULE: &'static str = "repository"; const ID: u32 = 6; - const CODEC_VERSION: u32 = 2; + const CODEC_VERSION: u32 = 3; type Input = CertificateInput; type Output = bool; fn execute( From 141faee02d91a038e4431a284c27ff6388282e56 Mon Sep 17 00:00:00 2001 From: forhappy Date: Fri, 2 Oct 2026 16:04:42 -0700 Subject: [PATCH 6/6] Release canonical pack reader ownership before cold clone qualification --- crates/canopy-server/tests/smart_http/main.rs | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/crates/canopy-server/tests/smart_http/main.rs b/crates/canopy-server/tests/smart_http/main.rs index 96c4f99..5b8aef7 100644 --- a/crates/canopy-server/tests/smart_http/main.rs +++ b/crates/canopy-server/tests/smart_http/main.rs @@ -103,7 +103,7 @@ async fn stock_git_push_and_clone_are_backed_by_one_repository_cell() )?; let repository = Arc::new(RepositoryCell::new( &application_handle, - target, + target.clone(), repository_id, canopy_server::ObjectFormat::Sha1, )?); @@ -441,12 +441,21 @@ async fn stock_git_push_and_clone_are_backed_by_one_repository_cell() }) .await?; drop(teardown_gateway); + // The canonical Cell also owns the shared pack reader. Drop both + // owners to prove complete cache teardown before the cold clone. + drop(repository); assert_eq!( disk_budget.used(), 0, - "gateway teardown releases retained object and snapshot charges" + "repository teardown releases retained object and snapshot charges" ); + let repository = Arc::new(RepositoryCell::new( + &application_handle, + target, + repository_id, + canopy_server::ObjectFormat::Sha1, + )?); let gateway = Arc::new(GitGateway::new( Arc::clone(&repository), scratch.path().to_path_buf(),