diff --git a/.github/actions/rust/action.yml b/.github/actions/rust/action.yml new file mode 100644 index 0000000..200508f --- /dev/null +++ b/.github/actions/rust/action.yml @@ -0,0 +1,38 @@ +name: Rust toolchain and cache +description: > + Install the toolchain rust-toolchain.toml names, and restore the cargo cache + for this job's workspaces. + +inputs: + shared-key: + description: > + Cache scope. Jobs sharing a key share one cache, so group jobs that build + the same artifacts and separate ones that do not. + required: true + workspaces: + description: > + Cargo workspaces to cache, one per line, as `path -> target`. The examples + are each their own workspace and are invisible to the default `.`. + required: false + default: "." + +runs: + using: composite + steps: + # rust-toolchain.toml is the single source of truth for the channel, the + # components and the wasm target. rustup honours it the moment cargo runs + # inside the repo, so `rustup show` installs exactly what it names — and a + # toolchain change stays a one-file edit rather than one edit per job that + # can silently drift from the file. + - run: rustup show + shell: bash + + - uses: Swatinem/rust-cache@v2 + with: + shared-key: ${{ inputs.shared-key }} + workspaces: ${{ inputs.workspaces }} + # Pull requests restore but never save. Four jobs each saving a + # multi-gigabyte cache is what pushes the repository past the 10 GB + # limit, and the eviction that follows throws away the entry the next + # run needed. main writes the cache every branch then restores. + save-if: ${{ github.ref == 'refs/heads/main' }} diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 19b1b51..927e9c2 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -6,99 +6,116 @@ on: pull_request: branches: [main] +# Two pushes to one pull request used to run two complete fans to completion, +# and only the second was ever read. +concurrency: + group: ci-${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + env: CARGO_TERM_COLOR: always RUST_BACKTRACE: 1 +# Four stages, each a checkout, a toolchain, and one script. The scripts are the +# recipes: they live in scripts/ so a job can be run locally with the same +# command CI uses, and so deploy/qemu-nitro/run-e2e.sh can share them instead of +# keeping a second copy that has to agree. +# +# There are deliberately no `needs:` edges. Gating the long jobs behind the +# short one would put its minutes on the front of the critical path to save +# runner time only on runs that were failing anyway; cancel-in-progress above +# removes the waste that actually dominates, which is superseded runs. jobs: - build: - name: build + test (default features) + unit: + name: fmt, clippy, unit tests runs-on: ubuntu-latest + timeout-minutes: 20 steps: - uses: actions/checkout@v4 - - name: Install Rust - uses: dtolnay/rust-toolchain@stable + - uses: ./.github/actions/rust with: - components: rustfmt, clippy - - uses: Swatinem/rust-cache@v2 - - name: cargo build - run: cargo build --workspace --all-targets - - name: cargo test (lib) - run: cargo test --workspace --lib - - name: cargo clippy - run: cargo clippy --workspace --all-targets -- -D warnings + shared-key: host + - run: scripts/ci-check.sh - build-aws: - name: build + test (aws feature) + bench: + name: benchmarks runs-on: ubuntu-latest + timeout-minutes: 45 + # Writing the tracked history back to gh-pages on main. + permissions: + contents: write steps: - uses: actions/checkout@v4 - - uses: dtolnay/rust-toolchain@stable + - uses: ./.github/actions/rust with: - components: clippy - - uses: Swatinem/rust-cache@v2 - - name: cargo build --features aws - run: cargo build --workspace --features aws --all-targets - - name: cargo test --features aws (lib) - run: cargo test --workspace --features aws --lib - - name: cargo clippy --features aws - run: cargo clippy --workspace --features aws --all-targets -- -D warnings - - minio-integration: - name: MinIO integration tests - runs-on: ubuntu-latest - services: - minio: - image: minio/minio:RELEASE.2025-02-28T09-55-16Z - ports: - - 9000:9000 - env: - MINIO_ROOT_USER: minioadmin - MINIO_ROOT_PASSWORD: minioadmin - options: >- - --health-cmd "curl -f http://localhost:9000/minio/health/ready" - --health-interval 5s - --health-timeout 3s - --health-retries 12 - ENTRYPOINT /usr/bin/docker-entrypoint.sh server /data - steps: - - uses: actions/checkout@v4 - - uses: dtolnay/rust-toolchain@stable - - uses: Swatinem/rust-cache@v2 - - name: cargo test (minio integration) - run: cargo test -p s3fs-core --features aws --test minio_integration -- --ignored --test-threads=1 + shared-key: guests + workspaces: | + . -> target + examples/guest-http -> target + - run: scripts/ci-bench.sh - fmt: - name: rustfmt - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - uses: dtolnay/rust-toolchain@stable + - uses: benchmark-action/github-action-benchmark@v1 with: - components: rustfmt - - run: cargo fmt --all -- --check + tool: cargo + output-file-path: bench.txt + github-token: ${{ secrets.GITHUB_TOKEN }} + # History is written only from main. A pull request compares against + # it and says so in the log, but must not rewrite the baseline. + auto-push: ${{ github.ref == 'refs/heads/main' }} + save-data-file: ${{ github.ref == 'refs/heads/main' }} + # A shared runner's neighbours move these numbers more than most + # commits do, so a regression is reported and never fails the build. + # Tightening this is worth doing once there is enough history to know + # what the noise floor actually is. + alert-threshold: "200%" + comment-on-alert: true + fail-on-alert: false - guest-build: - name: example guest builds for wasm32-wasip2 + # There is deliberately no SQLite job. The runtime measures its guest into + # PCR16 before it will serve, and that reads the NSM — which a hosted runner + # does not have, so the binary refuses to start there at all. The workload + # lives on in scripts/ci-sqlite.sh for a machine with an enclave. + e2e: + name: the whole stack, in an emulated enclave runs-on: ubuntu-latest + # The integration suites, then a cold Nix build of the image, then the + # emulated boot. This is the only integration signal, so it carries the + # suites that used to have their own job. + timeout-minutes: 150 steps: - uses: actions/checkout@v4 - - uses: dtolnay/rust-toolchain@stable + - uses: DeterminateSystems/nix-installer-action@main + # No Magic Nix Cache. It relays every store lookup through the Actions + # cache API, which rate-limits it, and Nix treats the refusals as fatal: + # the image build failed on "rate limit exceeded" with nothing wrong in + # it. nixpkgs comes from cache.nixos.org regardless; what is lost is only + # this repository's own derivations, rebuilt each run. + - uses: ./.github/actions/rust with: - targets: wasm32-wasip2 - - uses: Swatinem/rust-cache@v2 - - name: Install wasi-sdk - run: | - curl -sLO https://github.com/WebAssembly/wasi-sdk/releases/download/wasi-sdk-25/wasi-sdk-25.0-x86_64-linux.tar.gz - mkdir -p $HOME/wasi-sdk - tar xf wasi-sdk-25.0-x86_64-linux.tar.gz -C $HOME/wasi-sdk --strip-components=1 - - name: build guest-fsdemo (with bundled SQLite) + shared-key: guests + workspaces: | + . -> target + examples/guest-http -> target + examples/guest-grpc -> target + - uses: actions/cache@v4 + with: + path: target/qemu-nitro/tools + # The pinned version lives in the script, so the key follows the + # script rather than being a second place to update. + key: vhost-device-vsock-${{ hashFiles('scripts/ci-e2e.sh') }} + - run: scripts/ci-e2e.sh env: - CC_wasm32_wasip2: ${{ github.workspace }}/../wasi-sdk/bin/clang - AR_wasm32_wasip2: ${{ github.workspace }}/../wasi-sdk/bin/ar - run: | - export CC_wasm32_wasip2=$HOME/wasi-sdk/bin/clang - export AR_wasm32_wasip2=$HOME/wasi-sdk/bin/ar - export CFLAGS_wasm32_wasip2="--sysroot=$HOME/wasi-sdk/share/wasi-sysroot -DSQLITE_THREADSAFE=0 -DHAVE_USLEEP=1" - cd examples/guest-fsdemo - cargo build --release --target wasm32-wasip2 + # A runner boots the emulated enclave more slowly than a workstation, + # and the default is tuned for the latter. + TIMEOUT: 480 + # Nested virtualisation on hosted runners works but is not something + # GitHub supports, so this is the job most likely to fail for reasons + # outside this repository. Its logs are the only way to tell that apart + # from a real regression. + - uses: actions/upload-artifact@v4 + if: failure() + with: + name: e2e-logs + path: | + target/qemu-nitro/e2e/*.log + if-no-files-found: ignore diff --git a/.gitignore b/.gitignore index 61bfb9b..f1f0ffd 100644 --- a/.gitignore +++ b/.gitignore @@ -1,4 +1,6 @@ -/target +# Any crate's build directory, not only the workspace root: the example +# guests are separate workspaces with their own target/. +target/ **/*.rs.bk Cargo.lock.bak .env @@ -9,3 +11,16 @@ Cargo.lock.bak *.iml .DS_Store /wasm-cache/ + +# OpenTofu/Terraform working directory: vendored provider binaries, hundreds of +# MB, re-fetched by `tofu init`. The lock file IS committed — it pins provider +# versions and belongs under review. +deploy/tofu/.terraform/ + +# Run artifacts that scripts/ci-bench.sh and scripts/ci-sqlite.sh drop in the +# repository root. `bench.txt` exists only long enough for the benchmark tracker +# to read it, and `runtime.log` is the SQLite job's server output; both are +# rewritten by every run. Rooted with a leading slash so this covers the files +# those scripts write and not a log someone keeps deliberately elsewhere. +/bench.txt +/runtime.log diff --git a/Cargo.lock b/Cargo.lock index bd7651e..9d27025 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1,6 +1,6 @@ # This file is automatically @generated by Cargo. # It is not intended for manual editing. -version = 3 +version = 4 [[package]] name = "addr2line" @@ -121,6 +121,104 @@ version = "1.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c3d036a3c4ab069c7b410a2ce876bd74808d2d0888a82667669f8e783a898bf1" +[[package]] +name = "arc-swap" +version = "1.9.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c049c0be4daef0b145cb3555416b3b8ef5b7888a38aea1a3a155801fe7b0810b" +dependencies = [ + "rustversion", +] + +[[package]] +name = "arrayref" +version = "0.3.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "76a2e8124351fda1ef8aaaa3bbd7ebbcb486bbcd4225aca0aa0d84bb2db8fecb" + +[[package]] +name = "arrayvec" +version = "0.7.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3fb67a6e08acf24fdeccbac2cb6ac4305825bd1f117462e0e6f2f193345ad56" + +[[package]] +name = "asn1-rs" +version = "0.6.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5493c3bedbacf7fd7382c6346bbd66687d12bbaad3a89a2d2c303ee6cf20b048" +dependencies = [ + "asn1-rs-derive 0.5.1", + "asn1-rs-impl", + "displaydoc", + "nom", + "num-traits", + "rusticata-macros", + "thiserror 1.0.69", + "time", +] + +[[package]] +name = "asn1-rs" +version = "0.7.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b7f43a50ac4fdca5df8e885c21b835997f0a1cdee65494a6847694a98652d9d8" +dependencies = [ + "asn1-rs-derive 0.6.0", + "asn1-rs-impl", + "displaydoc", + "nom", + "num-traits", + "rusticata-macros", + "thiserror 2.0.18", + "time", +] + +[[package]] +name = "asn1-rs-derive" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "965c2d33e53cb6b267e148a4cb0760bc01f4904c1cd4bb4002a085bb016d1490" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", + "synstructure", +] + +[[package]] +name = "asn1-rs-derive" +version = "0.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3109e49b1e4909e9db6515a30c633684d68cdeaa252f215214cb4fa1a5bfee2c" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", + "synstructure", +] + +[[package]] +name = "asn1-rs-impl" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7b18050c2cd6fe86c3a76584ef5e0baf286d038cda203eb6223df2cc413565f7" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", +] + +[[package]] +name = "assert-json-diff" +version = "2.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "47e4f2b81832e72834d7518d8487a0396a28cc408186a2e8854c0f98011faf12" +dependencies = [ + "serde", + "serde_json", +] + [[package]] name = "astral-tokio-tar" version = "0.6.1" @@ -137,6 +235,60 @@ dependencies = [ "xattr", ] +[[package]] +name = "async-channel" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "924ed96dd52d1b75e9c1a3e6275715fd320f5f9439fb5a4a11fa51f4221158d2" +dependencies = [ + "concurrent-queue", + "event-listener-strategy", + "futures-core", + "pin-project-lite", +] + +[[package]] +name = "async-http-codec" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "096146020b08dbc4587685b0730a7ba905625af13c65f8028035cdfd69573c91" +dependencies = [ + "anyhow", + "futures", + "http 1.4.0", + "httparse", + "log", +] + +[[package]] +name = "async-io" +version = "2.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "456b8a8feb6f42d237746d4b3e9a178494627745c3c56c6ea55d92ba50d026fc" +dependencies = [ + "autocfg", + "cfg-if", + "concurrent-queue", + "futures-io", + "futures-lite", + "parking", + "polling", + "rustix 1.1.4", + "slab", + "windows-sys 0.61.2", +] + +[[package]] +name = "async-net" +version = "2.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b948000fad4873c1c9339d60f2623323a0cfd3816e5181033c6a5cb68b2accf7" +dependencies = [ + "async-io", + "blocking", + "futures-lite", +] + [[package]] name = "async-stream" version = "0.3.6" @@ -156,9 +308,15 @@ checksum = "c7c24de15d275a1ecfd47a380fb4d5ec9bfe0933f309ed5e705b775596a3574d" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] +[[package]] +name = "async-task" +version = "4.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b75356056920673b02621b35afd0f7dda9306d03c79a30f5c56c44cf256e3de" + [[package]] name = "async-trait" version = "0.1.89" @@ -167,7 +325,26 @@ checksum = "9035ad2d096bed7955a320ee7e2230574d28fd3c3a0f186cbea1ff3c7eed5dbb" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", +] + +[[package]] +name = "async-web-client" +version = "0.6.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8caf502b44d6d4be6154ac33af012cbb5fef11e6066edcfb42834217fbaf501b" +dependencies = [ + "async-http-codec", + "async-net", + "futures", + "futures-rustls", + "http 1.4.0", + "lazy_static", + "log", + "rustls-pki-types", + "serde", + "thiserror 1.0.69", + "webpki-roots 0.26.11", ] [[package]] @@ -192,8 +369,8 @@ dependencies = [ "aws-runtime", "aws-sdk-sts", "aws-smithy-async", - "aws-smithy-http", - "aws-smithy-json", + "aws-smithy-http 0.63.6", + "aws-smithy-json 0.62.5", "aws-smithy-runtime", "aws-smithy-runtime-api", "aws-smithy-types", @@ -209,9 +386,9 @@ dependencies = [ [[package]] name = "aws-credential-types" -version = "1.2.14" +version = "1.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f20799b373a1be121fe3005fba0c2090af9411573878f224df44b42727fcaf7" +checksum = "e93964ffdaf57857f544be3666a5f57570bb699e934700f11b49708f61bb556e" dependencies = [ "aws-smithy-async", "aws-smithy-runtime-api", @@ -219,17 +396,41 @@ dependencies = [ "zeroize", ] +[[package]] +name = "aws-lc-rs" +version = "1.18.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ce2b2dcc879c3bae0d371e77c99f2238400ef24ec001394befa67b6e543add9e" +dependencies = [ + "aws-lc-sys", + "untrusted 0.7.1", + "zeroize", +] + +[[package]] +name = "aws-lc-sys" +version = "0.44.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f09fae7be8bb3174e05c6afdb34199e6dc0c7c04ba9fa237b1967adfbde27483" +dependencies = [ + "cc", + "cmake", + "dunce", + "fs_extra", + "pkg-config", +] + [[package]] name = "aws-runtime" -version = "1.7.3" +version = "1.9.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5dcd93c82209ac7413532388067dce79be5a8780c1786e5fae3df22e4dee2864" +checksum = "c9007227e10b5fed2f3e0a2beff489211e2b5604c400b7a9d5d81ca9d64c24bb" dependencies = [ "aws-credential-types", "aws-sigv4", "aws-smithy-async", - "aws-smithy-eventstream", - "aws-smithy-http", + "aws-smithy-eventstream 0.61.2", + "aws-smithy-http 0.64.0", "aws-smithy-runtime", "aws-smithy-runtime-api", "aws-smithy-types", @@ -247,6 +448,57 @@ dependencies = [ "uuid", ] +[[package]] +name = "aws-sdk-cloudwatchlogs" +version = "1.149.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb9c35917d634db0fcf94f60a8c05752be8c755aafffc9adf5a24a337e317814" +dependencies = [ + "arc-swap", + "aws-credential-types", + "aws-runtime", + "aws-smithy-async", + "aws-smithy-eventstream 0.61.2", + "aws-smithy-http 0.64.0", + "aws-smithy-json 0.63.0", + "aws-smithy-observability 0.3.0", + "aws-smithy-runtime", + "aws-smithy-runtime-api", + "aws-smithy-schema", + "aws-smithy-types", + "aws-types", + "bytes", + "fastrand", + "http 0.2.12", + "http 1.4.0", + "regex-lite", + "tracing", +] + +[[package]] +name = "aws-sdk-kms" +version = "1.106.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2336350f96efcf9c2552b7fdb4dd07a0c1fef11f22b28fb020a9e33e7925cf9d" +dependencies = [ + "aws-credential-types", + "aws-runtime", + "aws-smithy-async", + "aws-smithy-http 0.63.6", + "aws-smithy-json 0.62.5", + "aws-smithy-observability 0.2.6", + "aws-smithy-runtime", + "aws-smithy-runtime-api", + "aws-smithy-types", + "aws-types", + "bytes", + "fastrand", + "http 0.2.12", + "http 1.4.0", + "regex-lite", + "tracing", +] + [[package]] name = "aws-sdk-s3" version = "1.131.0" @@ -258,10 +510,10 @@ dependencies = [ "aws-sigv4", "aws-smithy-async", "aws-smithy-checksums", - "aws-smithy-eventstream", - "aws-smithy-http", - "aws-smithy-json", - "aws-smithy-observability", + "aws-smithy-eventstream 0.60.20", + "aws-smithy-http 0.63.6", + "aws-smithy-json 0.62.5", + "aws-smithy-observability 0.2.6", "aws-smithy-runtime", "aws-smithy-runtime-api", "aws-smithy-types", @@ -282,6 +534,30 @@ dependencies = [ "url", ] +[[package]] +name = "aws-sdk-ssm" +version = "1.109.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "69f4bdbeea2c7d18632093cd158644902f1e91ae025a3f68afaa449f620ae658" +dependencies = [ + "aws-credential-types", + "aws-runtime", + "aws-smithy-async", + "aws-smithy-http 0.63.6", + "aws-smithy-json 0.62.5", + "aws-smithy-observability 0.2.6", + "aws-smithy-runtime", + "aws-smithy-runtime-api", + "aws-smithy-types", + "aws-types", + "bytes", + "fastrand", + "http 0.2.12", + "http 1.4.0", + "regex-lite", + "tracing", +] + [[package]] name = "aws-sdk-sts" version = "1.103.0" @@ -291,9 +567,9 @@ dependencies = [ "aws-credential-types", "aws-runtime", "aws-smithy-async", - "aws-smithy-http", - "aws-smithy-json", - "aws-smithy-observability", + "aws-smithy-http 0.63.6", + "aws-smithy-json 0.62.5", + "aws-smithy-observability 0.2.6", "aws-smithy-query", "aws-smithy-runtime", "aws-smithy-runtime-api", @@ -309,13 +585,13 @@ dependencies = [ [[package]] name = "aws-sigv4" -version = "1.4.3" +version = "1.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68dc0b907359b120170613b5c09ccc61304eac3998ff6274b97d93ee6490115a" +checksum = "723c2234ad7511ceef63eab016b7ba6ff7c55590fefb96fa8467af014a07309f" dependencies = [ "aws-credential-types", - "aws-smithy-eventstream", - "aws-smithy-http", + "aws-smithy-eventstream 0.61.2", + "aws-smithy-http 0.64.0", "aws-smithy-runtime-api", "aws-smithy-types", "bytes", @@ -332,9 +608,9 @@ dependencies = [ [[package]] name = "aws-smithy-async" -version = "1.2.14" +version = "1.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2ffcaf626bdda484571968400c326a244598634dc75fd451325a54ad1a59acfc" +checksum = "f02e407fb3b54891734224b9ffac8a71fdd35f542500fa1af95754a6b2beb316" dependencies = [ "futures-util", "pin-project-lite", @@ -347,7 +623,7 @@ version = "0.64.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "10efbbcec1e044b81600e2fc562a391951d291152d95b482d5b7e7132299d762" dependencies = [ - "aws-smithy-http", + "aws-smithy-http 0.63.6", "aws-smithy-types", "bytes", "crc-fast", @@ -373,13 +649,46 @@ dependencies = [ "crc32fast", ] +[[package]] +name = "aws-smithy-eventstream" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6de526c7b567420a31bc283657a7921b45c4cafe0827fdf2490713dcc770c28f" +dependencies = [ + "aws-smithy-types", + "bytes", + "crc32fast", +] + [[package]] name = "aws-smithy-http" version = "0.63.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ba1ab2dc1c2c3749ead27180d333c42f11be8b0e934058fb4b2258ee8dbe5231" dependencies = [ - "aws-smithy-eventstream", + "aws-smithy-eventstream 0.60.20", + "aws-smithy-runtime-api", + "aws-smithy-types", + "bytes", + "bytes-utils", + "futures-core", + "futures-util", + "http 1.4.0", + "http-body 1.0.1", + "http-body-util", + "percent-encoding", + "pin-project-lite", + "pin-utils", + "tracing", +] + +[[package]] +name = "aws-smithy-http" +version = "0.64.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "37843d9add67c3aff5856f409c6dc315d3cdff60f9c0cb5b670dab1e9920306d" +dependencies = [ + "aws-smithy-eventstream 0.61.2", "aws-smithy-runtime-api", "aws-smithy-types", "bytes", @@ -397,23 +706,37 @@ dependencies = [ [[package]] name = "aws-smithy-http-client" -version = "1.1.12" +version = "1.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6a2f165a7feee6f263028b899d0a181987f4fa7179a6411a32a439fba7c5f769" +checksum = "ebfd138fac0337cee7516c352757ea73b9f2266e57d0bcb5bc70e9547e45aef1" dependencies = [ "aws-smithy-async", + "aws-smithy-protocol-test", "aws-smithy-runtime-api", "aws-smithy-types", + "bytes", "h2 0.3.27", "h2 0.4.13", "http 0.2.12", + "http 1.4.0", "http-body 0.4.6", + "http-body 1.0.1", "hyper 0.14.32", + "hyper 1.9.0", "hyper-rustls 0.24.2", + "hyper-rustls 0.27.9", + "hyper-util", + "indexmap 2.14.0", "pin-project-lite", "rustls 0.21.12", + "rustls 0.23.40", "rustls-native-certs", + "rustls-pki-types", + "serde", + "serde_json", "tokio", + "tokio-rustls 0.26.4", + "tower", "tracing", ] @@ -426,6 +749,17 @@ dependencies = [ "aws-smithy-types", ] +[[package]] +name = "aws-smithy-json" +version = "0.63.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3dc65a121adb4b33729919fcfa14fa36fb33c1555a8f06bb0e2188dbfdc1d9ef" +dependencies = [ + "aws-smithy-runtime-api", + "aws-smithy-schema", + "aws-smithy-types", +] + [[package]] name = "aws-smithy-observability" version = "0.2.6" @@ -435,6 +769,34 @@ dependencies = [ "aws-smithy-runtime-api", ] +[[package]] +name = "aws-smithy-observability" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e86338c869539a581bf161247762a6e87f92c5c075060057b5ed6d06632ed0c" +dependencies = [ + "aws-smithy-runtime-api", +] + +[[package]] +name = "aws-smithy-protocol-test" +version = "0.64.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f76511a0e223ce78deb6a78b8afebda99cb737cfbc8a58d96dcb190f012dd40a" +dependencies = [ + "assert-json-diff", + "aws-smithy-runtime-api", + "base64-simd", + "cbor-diag", + "ciborium", + "http 0.2.12", + "pretty_assertions", + "regex-lite", + "roxmltree", + "serde_json", + "thiserror 2.0.18", +] + [[package]] name = "aws-smithy-query" version = "0.60.15" @@ -447,15 +809,16 @@ dependencies = [ [[package]] name = "aws-smithy-runtime" -version = "1.11.1" +version = "1.14.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0504b1ab12debb5959e5165ee5fe97dd387e7aa7ea6a477bfd7635dfe769a4f5" +checksum = "b82e438d30e02a825d363bd639a9efaed68a8089d86101054b0081e7e0d3e606" dependencies = [ "aws-smithy-async", - "aws-smithy-http", + "aws-smithy-http 0.64.0", "aws-smithy-http-client", - "aws-smithy-observability", + "aws-smithy-observability 0.3.0", "aws-smithy-runtime-api", + "aws-smithy-schema", "aws-smithy-types", "bytes", "fastrand", @@ -472,9 +835,9 @@ dependencies = [ [[package]] name = "aws-smithy-runtime-api" -version = "1.12.0" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b71a13df6ada0aafbf21a73bdfcdf9324cfa9df77d96b8446045be3cde61b42e" +checksum = "954c563ce84507722d2679f07a35d21b9c6466b3872d513020d0281fc8112ac9" dependencies = [ "aws-smithy-async", "aws-smithy-runtime-api-macros", @@ -490,20 +853,31 @@ dependencies = [ [[package]] name = "aws-smithy-runtime-api-macros" -version = "1.0.0" +version = "1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8d7396fd9500589e62e460e987ecb671bad374934e55ec3b5f498cc7a8a8a7b7" +checksum = "221eaa237ddf1ca79b60d1372aad77e47f9c0ea5b3ce5099da8c61d027dc77b3" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", +] + +[[package]] +name = "aws-smithy-schema" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7d56e0a4e53127a632224e43633b0fe045fa9e1e3cfc68b9830f1115e103f910" +dependencies = [ + "aws-smithy-runtime-api", + "aws-smithy-types", + "http 1.4.0", ] [[package]] name = "aws-smithy-types" -version = "1.4.7" +version = "1.6.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d73dbfbaa8e4bc57b9045137680b958d274823509a360abfd8e1d514d40c95c" +checksum = "fce83ce9abbb198d25bc7131e468d0f9fe1257125e58c39f3f9fc9f5098c9647" dependencies = [ "base64-simd", "bytes", @@ -536,13 +910,14 @@ dependencies = [ [[package]] name = "aws-types" -version = "1.3.15" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f4bbcaa9304ea40902d3d5f42a0428d1bd895a2b0f6999436fb279ffddc58ac" +checksum = "eec1cd5469f328c782dc3e33d4153cf118a54e33cbb3356d60d16f89883e1f94" dependencies = [ "aws-credential-types", "aws-smithy-async", "aws-smithy-runtime-api", + "aws-smithy-schema", "aws-smithy-types", "rustc_version", "tracing", @@ -591,6 +966,12 @@ dependencies = [ "tower-service", ] +[[package]] +name = "base64" +version = "0.21.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9d297deb1925b89f2ccc13d7635fa0714f12c87adce1c75356b39ca9b7178567" + [[package]] name = "base64" version = "0.22.1" @@ -607,6 +988,32 @@ dependencies = [ "vsimd", ] +[[package]] +name = "base64ct" +version = "1.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2af50177e190e07a26ab74f8b1efbfe2ef87da2116221318cb1c2e82baf7de06" + +[[package]] +name = "base64urlsafedata" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b08e33815c87d8cadcddb1e74ac307368a3751fbe40c961538afa21a1899f21c" +dependencies = [ + "base64 0.21.7", + "pastey", + "serde", +] + +[[package]] +name = "bit-vec" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b71798fca2c1fe1086445a7258a4bc81e6e49dcd24c8d0dd9a1e57395b603f51" +dependencies = [ + "serde", +] + [[package]] name = "bitflags" version = "2.11.1" @@ -625,6 +1032,20 @@ dependencies = [ "wyz", ] +[[package]] +name = "blake3" +version = "1.8.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "76ae7bad254120e9e4c63bafc385310756f90c484eac0e36b8317cf09cb92a77" +dependencies = [ + "arrayref", + "arrayvec", + "cc", + "cfg-if", + "constant_time_eq", + "cpufeatures 0.3.0", +] + [[package]] name = "block-buffer" version = "0.10.4" @@ -643,6 +1064,19 @@ dependencies = [ "hybrid-array", ] +[[package]] +name = "blocking" +version = "1.6.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e83f8d02be6967315521be875afa792a316e28d57b5a2d401897e2a7921b7f21" +dependencies = [ + "async-channel", + "async-task", + "futures-io", + "futures-lite", + "piper", +] + [[package]] name = "bollard" version = "0.20.2" @@ -650,7 +1084,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ee04c4c84f1f811b017f2fbb7dd8815c976e7ca98593de9c1e2afad0f636bff4" dependencies = [ "async-stream", - "base64", + "base64 0.22.1", "bitflags", "bollard-buildkit-proto", "bollard-stubs", @@ -707,7 +1141,7 @@ version = "1.52.1-rc.29.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0f0a8ca8799131c1837d1282c3f81f31e76ceb0ce426e04a7fe1ccee3287c066" dependencies = [ - "base64", + "base64 0.22.1", "bollard-buildkit-proto", "bytes", "prost", @@ -717,6 +1151,15 @@ dependencies = [ "time", ] +[[package]] +name = "bs58" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bf88ba1141d185c399bee5288d850d63b8369520c1eafc32a0430b5b6c287bf4" +dependencies = [ + "tinyvec", +] + [[package]] name = "bumpalo" version = "3.20.2" @@ -826,6 +1269,25 @@ version = "0.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5" +[[package]] +name = "cbor-diag" +version = "0.1.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc245b6ecd09b23901a4fbad1ad975701fd5061ceaef6afa93a2d70605a64429" +dependencies = [ + "bs58", + "chrono", + "data-encoding", + "half", + "nom", + "num-bigint", + "num-rational", + "num-traits", + "separator", + "url", + "uuid", +] + [[package]] name = "cc" version = "1.2.61" @@ -925,7 +1387,7 @@ dependencies = [ "heck", "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -934,12 +1396,33 @@ version = "1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c8d4a3bb8b1e0c1050499d1815f5ab16d04f0959b233085fb31653fbfc9d98f9" +[[package]] +name = "cmake" +version = "0.1.58" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c0f78a02292a74a88ac736019ab962ece0bc380e3f977bf72e376c5d78ff0678" +dependencies = [ + "cc", +] + [[package]] name = "cmov" version = "0.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3f88a43d011fc4a6876cb7344703e297c71dda42494fee094d5f7c76bf13f746" +[[package]] +name = "cms" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7b77c319abfd5219629c45c34c89ba945ed3c5e49fcde9d16b6c3885f118a730" +dependencies = [ + "const-oid 0.9.6", + "der", + "spki", + "x509-cert", +] + [[package]] name = "cobs" version = "0.3.0" @@ -955,12 +1438,33 @@ version = "1.0.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" +[[package]] +name = "concurrent-queue" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4ca0197aee26d1ae37445ee532fefce43251d24cc7c166799f4d46817f1d3973" +dependencies = [ + "crossbeam-utils", +] + +[[package]] +name = "const-oid" +version = "0.9.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c2459377285ad874054d797f3ccebf984978aa39129f6eafde5cdc8315b612f8" + [[package]] name = "const-oid" version = "0.10.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a6ef517f0926dd24a1582492c791b6a4818a4d94e789a334894aa15b0d12f55c" +[[package]] +name = "constant_time_eq" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b" + [[package]] name = "core-foundation" version = "0.10.1" @@ -977,6 +1481,16 @@ version = "0.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" +[[package]] +name = "coset" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1eb98d5e9155e2cf7cd942c8b3033097d4563b6fb0a00b9caecb74669555c058" +dependencies = [ + "ciborium", + "ciborium-io", +] + [[package]] name = "cpp_demangle" version = "0.4.5" @@ -1299,7 +1813,7 @@ dependencies = [ "proc-macro2", "quote", "strsim", - "syn", + "syn 2.0.117", ] [[package]] @@ -1310,9 +1824,15 @@ checksum = "ac3984ec7bd6cfa798e62b4a642426a5be0e68f9401cfc2a01e3fa9ea2fcdb8d" dependencies = [ "darling_core", "quote", - "syn", + "syn 2.0.117", ] +[[package]] +name = "data-encoding" +version = "2.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4583a4551df46e2792f82ceeac45e850d2e2d5debba0b91f102385cda5b11f06" + [[package]] name = "debugid" version = "0.8.0" @@ -1322,6 +1842,58 @@ dependencies = [ "uuid", ] +[[package]] +name = "der" +version = "0.7.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7c1832837b905bbfb5101e07cc24c8deddf52f93225eee6ead5f4d63d53ddcb" +dependencies = [ + "const-oid 0.9.6", + "der_derive", + "flagset", + "pem-rfc7468", + "zeroize", +] + +[[package]] +name = "der-parser" +version = "9.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5cd0a5c643689626bec213c4d8bd4d96acc8ffdb4ad4bb6bc16abf27d5f4b553" +dependencies = [ + "asn1-rs 0.6.2", + "displaydoc", + "nom", + "num-bigint", + "num-traits", + "rusticata-macros", +] + +[[package]] +name = "der-parser" +version = "10.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "07da5016415d5a3c4dd39b11ed26f915f52fc4e0dc197d87908bc916e51bc1a6" +dependencies = [ + "asn1-rs 0.7.2", + "displaydoc", + "nom", + "num-bigint", + "num-traits", + "rusticata-macros", +] + +[[package]] +name = "der_derive" +version = "0.7.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8034092389675178f570469e6c3b0465d3d30b4505c294a6550db47f3c17ad18" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", +] + [[package]] name = "deranged" version = "0.5.8" @@ -1332,6 +1904,12 @@ dependencies = [ "serde_core", ] +[[package]] +name = "diff" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "56254986775e3233ffa9c4d7d3faaf6d36a2c09d30b20687e9f88bc8bafc16c8" + [[package]] name = "digest" version = "0.10.7" @@ -1349,7 +1927,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4850db49bf08e663084f7fb5c87d202ef91a3907271aff24a94eb97ff039153c" dependencies = [ "block-buffer 0.12.0", - "const-oid", + "const-oid 0.10.2", "crypto-common 0.2.1", "ctutils", ] @@ -1383,7 +1961,7 @@ checksum = "97369cbbc041bc366949bc74d34658d6cda5621039731c6310521892a3a20ae0" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -1392,11 +1970,17 @@ version = "1.3.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a4564c274ebf369f501de192b02a0b81a5c4bda375abfe526aa70fc702fa6fa0" dependencies = [ - "base64", + "base64 0.22.1", "serde", "serde_json", ] +[[package]] +name = "dunce" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813" + [[package]] name = "dyn-clone" version = "1.0.20" @@ -1421,6 +2005,69 @@ version = "0.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "edd0f118536f44f5ccd48bcb8b111bdc3de888b58c74639dfb034a357d0f206d" +[[package]] +name = "enclave-runtime" +version = "0.1.0" +dependencies = [ + "anyhow", + "async-trait", + "aws-config", + "aws-lc-rs", + "aws-sdk-cloudwatchlogs", + "aws-sdk-kms", + "aws-sdk-ssm", + "aws-smithy-http-client", + "aws-smithy-types", + "base64 0.22.1", + "blake3", + "bytes", + "cap-rand", + "ciborium", + "clap", + "cms", + "criterion", + "der", + "enclave-runtime", + "futures", + "getrandom 0.3.4", + "hex", + "http 1.4.0", + "http-body-util", + "hyper 1.9.0", + "hyper-rustls 0.27.9", + "hyper-util", + "nitro-attestation", + "nitro-nsm", + "pem", + "prost", + "rcgen 0.14.8", + "rustix 1.1.4", + "rustls 0.23.40", + "rustls-acme", + "rustls-webpki 0.103.13", + "s3fs-core", + "serde", + "serde_json", + "spki", + "tokio", + "tokio-rustls 0.26.4", + "tokio-stream", + "tonic", + "tonic-prost", + "tower", + "tracing", + "tracing-subscriber", + "wasmtime", + "wasmtime-wasi", + "wasmtime-wasi-http", + "wasmtime-wasi-io", + "webauthn-rs", + "webpki-roots 1.0.9", + "x509-cert", + "x509-parser 0.18.1", + "zeroize", +] + [[package]] name = "encoding_rs" version = "0.8.35" @@ -1456,6 +2103,26 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "event-listener" +version = "5.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" +dependencies = [ + "parking", + "pin-project-lite", +] + +[[package]] +name = "event-listener-strategy" +version = "0.5.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8be9f3dfaaffdae2972880079a491a1a8bb7cbed0b8dd7a347f668b4150a3b93" +dependencies = [ + "event-listener", + "pin-project-lite", +] + [[package]] name = "fastrand" version = "2.4.1" @@ -1507,6 +2174,12 @@ version = "0.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0ce7134b9999ecaf8bcd65542e436736ef32ddca1b3e06094cb6ec5755203b80" +[[package]] +name = "flagset" +version = "0.4.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b7ac824320a75a52197e8f2d787f6a38b6718bb6897a35142d749af3c0e8f4fe" + [[package]] name = "fnv" version = "1.0.7" @@ -1525,6 +2198,21 @@ version = "0.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "77ce24cb58228fbb8aa041425bb1050850ac19177686ea6e0f41a70416f56fdb" +[[package]] +name = "foreign-types" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f6f339eb8adc052cd2ca78910fda869aefa38d22d5cb648e6485e4d3fc06f3b1" +dependencies = [ + "foreign-types-shared", +] + +[[package]] +name = "foreign-types-shared" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "00b0228411908ca8685dba7fc2cdd70ec9990a6e753e89b6ac91a84c40fbaf4b" + [[package]] name = "form_urlencoded" version = "1.2.2" @@ -1545,6 +2233,12 @@ dependencies = [ "windows-sys 0.59.0", ] +[[package]] +name = "fs_extra" +version = "1.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" + [[package]] name = "funty" version = "2.0.0" @@ -1599,6 +2293,19 @@ version = "0.3.32" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718" +[[package]] +name = "futures-lite" +version = "2.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f78e10609fe0e0b3f4157ffab1876319b5b0db102a2c60dc4626306dc46b44ad" +dependencies = [ + "fastrand", + "futures-core", + "futures-io", + "parking", + "pin-project-lite", +] + [[package]] name = "futures-macro" version = "0.3.32" @@ -1607,7 +2314,18 @@ checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", +] + +[[package]] +name = "futures-rustls" +version = "0.26.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a8f2f12607f92c69b12ed746fabf9ca4f5c482cba46679c1a75b874ed7c26adb" +dependencies = [ + "futures-io", + "rustls 0.23.40", + "rustls-pki-types", ] [[package]] @@ -2014,6 +2732,7 @@ dependencies = [ "hyper 1.9.0", "hyper-util", "rustls 0.23.40", + "rustls-native-certs", "tokio", "tokio-rustls 0.26.4", "tower-service", @@ -2038,13 +2757,16 @@ version = "0.1.20" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "96547c2556ec9d12fb1578c4eaf448b04993e7fb79cbaad930a656880a6bdfa0" dependencies = [ + "base64 0.22.1", "bytes", "futures-channel", "futures-util", "http 1.4.0", "http-body 1.0.1", "hyper 1.9.0", + "ipnet", "libc", + "percent-encoding", "pin-project-lite", "socket2 0.6.3", "tokio", @@ -2479,6 +3201,12 @@ version = "0.3.17" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "6877bb514081ee2a7ff5ef9de3281f14a4dd4bceac4c09388074a6b5df8a139a" +[[package]] +name = "minimal-lexical" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "68354c5c6bd36d73ff3feceb05efa59b6acb7626617f4962be322a825e61f79a" + [[package]] name = "mio" version = "1.2.0" @@ -2490,6 +3218,44 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "nitro-attestation" +version = "0.1.0" +dependencies = [ + "anyhow", + "aws-lc-rs", + "base64 0.22.1", + "ciborium", + "clap", + "coset", + "hex", + "rcgen 0.14.8", + "rustls 0.23.40", + "rustls-pki-types", + "time", + "x509-parser 0.18.1", +] + +[[package]] +name = "nitro-nsm" +version = "0.1.0" +dependencies = [ + "anyhow", + "aws-lc-rs", + "ciborium", + "rustix 1.1.4", +] + +[[package]] +name = "nom" +version = "7.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d273983c5a657a70a3e8f2a01329822f3b8c8172b73826411a55751e404a0a4a" +dependencies = [ + "memchr", + "minimal-lexical", +] + [[package]] name = "nu-ansi-term" version = "0.50.3" @@ -2590,6 +3356,24 @@ dependencies = [ "memchr", ] +[[package]] +name = "oid-registry" +version = "0.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a8d8034d9489cdaf79228eb9f6a3b8d7bb32ba00d6645ebd48eef4077ceb5bd9" +dependencies = [ + "asn1-rs 0.6.2", +] + +[[package]] +name = "oid-registry" +version = "0.8.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "12f40cff3dde1b6087cc5d5f5d4d65712f34016a03ed60e9c08dcc392736b5b7" +dependencies = [ + "asn1-rs 0.7.2", +] + [[package]] name = "once_cell" version = "1.21.4" @@ -2608,18 +3392,61 @@ version = "11.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d6790f58c7ff633d8771f42965289203411a5e5c68388703c06e14f24770b41e" +[[package]] +name = "openssl" +version = "0.10.81" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "77823a27f0babb03091cb9ed9ef80af3b39dbc82f97e8fa530374b7dafd87a45" +dependencies = [ + "bitflags", + "cfg-if", + "foreign-types", + "libc", + "openssl-macros", + "openssl-sys", +] + +[[package]] +name = "openssl-macros" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a948666b637a0f465e8564c73e89d4dde00d72d4d473cc972f390fc3dcee7d9c" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", +] + [[package]] name = "openssl-probe" version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7c87def4c32ab89d880effc9e097653c8da5d6ef28e6b539d313baaacfbafcbe" +[[package]] +name = "openssl-sys" +version = "0.9.117" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b47e7e6bb2c38cd930d25a23b40fa52e068c10e85f3e03a7f5ba5aaca5713695" +dependencies = [ + "cc", + "libc", + "pkg-config", + "vcpkg", +] + [[package]] name = "outref" version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1a80800c0488c3a21695ea981a54918fbb37abf04f4d0720c453632255e2ff0e" +[[package]] +name = "parking" +version = "2.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f38d5652c16fde515bb1ecef450ab0f6a219d619a7274976324d5e377f7dceba" + [[package]] name = "parking_lot" version = "0.12.5" @@ -2665,7 +3492,32 @@ dependencies = [ "regex", "regex-syntax", "structmeta", - "syn", + "syn 2.0.117", +] + +[[package]] +name = "pastey" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "35fb2e5f958ec131621fdd531e9fc186ed768cbe395337403ae56c17a74c68ec" + +[[package]] +name = "pem" +version = "3.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d30c53c26bc5b31a98cd02d20f25a7c8567146caf63ed593a9d87b2775291be" +dependencies = [ + "base64 0.22.1", + "serde_core", +] + +[[package]] +name = "pem-rfc7468" +version = "0.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "88b39c9bfcfc231068454382784bb460aae594343fb030d46e9f50a645418412" +dependencies = [ + "base64ct", ] [[package]] @@ -2701,7 +3553,7 @@ checksum = "d9b20ed30f105399776b9c883e68e536ef602a16ae6f596d2c473591d6ad64c6" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -2716,6 +3568,17 @@ version = "0.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8b870d8c151b6f2fb93e84a13146138f05d02ed11c7e7c54f8826aaaf7c9f184" +[[package]] +name = "piper" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c835479a4443ded371d6c535cbfd8d31ad92c5d23ae9770a61bc155e4992a3c1" +dependencies = [ + "atomic-waker", + "fastrand", + "futures-io", +] + [[package]] name = "pkg-config" version = "0.3.33" @@ -2756,6 +3619,20 @@ dependencies = [ "plotters-backend", ] +[[package]] +name = "polling" +version = "3.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5d0e4f59085d47d8241c88ead0f274e8a0cb551f3625263c05eb8dd897c34218" +dependencies = [ + "cfg-if", + "concurrent-queue", + "hermit-abi", + "pin-project-lite", + "rustix 1.1.4", + "windows-sys 0.61.2", +] + [[package]] name = "portable-atomic" version = "1.13.1" @@ -2790,12 +3667,22 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "439ee305def115ba05938db6eb1644ff94165c5ab5e9420d1c1bcedbba909391" [[package]] -name = "ppv-lite86" -version = "0.2.21" +name = "ppv-lite86" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" +dependencies = [ + "zerocopy", +] + +[[package]] +name = "pretty_assertions" +version = "1.4.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" +checksum = "3ae130e2f271fbc2ac3a40fb1d07180839cdbbe443c7a27e1e3c13c5cac0116d" dependencies = [ - "zerocopy", + "diff", + "yansi", ] [[package]] @@ -2805,7 +3692,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "479ca8adacdd7ce8f1fb39ce9ecccbfe93a3f1344b3d0d97f20bc0196208f62b" dependencies = [ "proc-macro2", - "syn", + "syn 2.0.117", ] [[package]] @@ -2837,7 +3724,7 @@ dependencies = [ "itertools 0.14.0", "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -2869,7 +3756,7 @@ checksum = "f7dfa8354acc622b3857e1bb1a4e4315d3bc1a44ad31d5653c3e87c0da9306d7" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -2995,6 +3882,33 @@ dependencies = [ "crossbeam-utils", ] +[[package]] +name = "rcgen" +version = "0.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "75e669e5202259b5314d1ea5397316ad400819437857b90861765f24c4cf80a2" +dependencies = [ + "aws-lc-rs", + "pem", + "rustls-pki-types", + "time", + "yasna 0.5.2", +] + +[[package]] +name = "rcgen" +version = "0.14.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "57f6d249aad744e274e682777a50283a225a32705394ee6d5fcc01efa25e4055" +dependencies = [ + "aws-lc-rs", + "pem", + "rustls-pki-types", + "time", + "x509-parser 0.18.1", + "yasna 0.6.0", +] + [[package]] name = "redox_syscall" version = "0.5.18" @@ -3041,7 +3955,7 @@ checksum = "b7186006dcb21920990093f30e3dea63b7d6e977bf1256be20c3563a5db070da" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -3103,10 +4017,19 @@ dependencies = [ "cfg-if", "getrandom 0.2.17", "libc", - "untrusted", + "untrusted 0.9.0", "windows-sys 0.52.0", ] +[[package]] +name = "roxmltree" +version = "0.14.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "921904a62e410e37e215c40381b7117f830d9d89ba60ab5236170541dd25646b" +dependencies = [ + "xmlparser", +] + [[package]] name = "rustc-demangle" version = "0.1.27" @@ -3128,6 +4051,15 @@ dependencies = [ "semver", ] +[[package]] +name = "rusticata-macros" +version = "4.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "faf0c4a6ece9950b9abdb62b1cfcf2a68b3b67a10ba445b3bb85be2a293d0632" +dependencies = [ + "nom", +] + [[package]] name = "rustix" version = "0.38.44" @@ -3182,6 +4114,7 @@ version = "0.23.40" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ef86cd5876211988985292b91c96a8f2d298df24e75989a43a3c73f2d4d8168b" dependencies = [ + "aws-lc-rs", "log", "once_cell", "ring", @@ -3191,6 +4124,34 @@ dependencies = [ "zeroize", ] +[[package]] +name = "rustls-acme" +version = "0.15.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9c70a17ecb067d5067565a16a2e0f26a4a2ea0924f49739d558c45186facc75" +dependencies = [ + "async-io", + "async-trait", + "async-web-client", + "aws-lc-rs", + "base64 0.22.1", + "blocking", + "chrono", + "futures", + "futures-rustls", + "http 1.4.0", + "log", + "pem", + "rcgen 0.13.2", + "serde", + "serde_json", + "thiserror 2.0.18", + "tokio", + "tokio-util", + "webpki-roots 1.0.9", + "x509-parser 0.16.0", +] + [[package]] name = "rustls-native-certs" version = "0.8.3" @@ -3219,7 +4180,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8b6275d1ee7a1cd780b64aca7726599a1dbc893b1e64144529e55c3c2f745765" dependencies = [ "ring", - "untrusted", + "untrusted 0.9.0", ] [[package]] @@ -3228,9 +4189,10 @@ version = "0.103.13" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" dependencies = [ + "aws-lc-rs", "ring", "rustls-pki-types", - "untrusted", + "untrusted 0.9.0", ] [[package]] @@ -3252,10 +4214,12 @@ dependencies = [ "anyhow", "async-trait", "aws-config", + "aws-lc-rs", "aws-sdk-s3", "aws-smithy-runtime-api", "aws-smithy-types", "bitvec", + "blake3", "bytes", "criterion", "fastrand", @@ -3270,37 +4234,7 @@ dependencies = [ "tokio", "tokio-util", "tracing", -] - -[[package]] -name = "s3fs-runner" -version = "0.1.0" -dependencies = [ - "anyhow", - "clap", - "s3fs-core", - "s3fs-wasmtime", - "tokio", - "tracing", - "tracing-subscriber", - "wasmtime", - "wasmtime-wasi", - "wasmtime-wasi-io", -] - -[[package]] -name = "s3fs-wasmtime" -version = "0.1.0" -dependencies = [ - "anyhow", - "async-trait", - "bytes", - "s3fs-core", - "tokio", - "tracing", - "wasmtime", - "wasmtime-wasi", - "wasmtime-wasi-io", + "zeroize", ] [[package]] @@ -3358,7 +4292,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "da046153aa2352493d6cb7da4b6e5c0c057d8a1d0a9aa8560baffdd945acd414" dependencies = [ "ring", - "untrusted", + "untrusted 0.9.0", ] [[package]] @@ -3394,42 +4328,59 @@ dependencies = [ "serde_core", ] +[[package]] +name = "separator" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f97841a747eef040fcd2e7b3b9a220a7205926e60488e673d9e4926d27772ce5" + [[package]] name = "serde" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" dependencies = [ "serde_core", "serde_derive", ] +[[package]] +name = "serde_cbor_2" +version = "0.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34aec2709de9078e077090abd848e967abab63c9fb3fdb5d4799ad359d8d482c" +dependencies = [ + "half", + "serde", +] + [[package]] name = "serde_core" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" dependencies = [ "serde_derive", ] [[package]] name = "serde_derive" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] name = "serde_json" -version = "1.0.149" +version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "83fc039473c5595ace860d8c4fafa220ff474b3fc6bfdb4293327f1a37e94d86" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ + "indexmap 2.14.0", "itoa", "memchr", "serde", @@ -3445,7 +4396,7 @@ checksum = "175ee3e80ae9982737ca543e96133087cbd9a485eecc3bc4de9c1a37b47ea59c" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -3475,7 +4426,7 @@ version = "3.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "dd5414fad8e6907dbdd5bc441a50ae8d6e26151a03b1de04d89a5576de61d01f" dependencies = [ - "base64", + "base64 0.22.1", "chrono", "hex", "indexmap 1.9.3", @@ -3497,7 +4448,7 @@ dependencies = [ "darling", "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -3605,6 +4556,16 @@ version = "0.10.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d5fe4ccb98d9c292d56fec89a5e07da7fc4cf0dc11e156b41793132775d3e591" +[[package]] +name = "spki" +version = "0.7.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d91ed6c858b01f942cd56b37a94b3e0a1798290327d1236e4d9cf4eaca44d29d" +dependencies = [ + "base64ct", + "der", +] + [[package]] name = "stable_deref_trait" version = "1.2.1" @@ -3626,7 +4587,7 @@ dependencies = [ "proc-macro2", "quote", "structmeta-derive", - "syn", + "syn 2.0.117", ] [[package]] @@ -3637,7 +4598,7 @@ checksum = "152a0b65a590ff6c3da95cabe2353ee04e6167c896b28e3b14478c2636c922fc" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -3657,6 +4618,17 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "syn" +version = "3.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + [[package]] name = "sync_wrapper" version = "1.0.2" @@ -3671,7 +4643,7 @@ checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -3790,7 +4762,7 @@ checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -3801,7 +4773,7 @@ checksum = "ebc4ee7f67670e9b64d05fa4253e753e016c6c95ff35b89b7941d6b856dec1d5" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -3864,6 +4836,42 @@ dependencies = [ "serde_json", ] +[[package]] +name = "tinyvec" +version = "1.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb4ebadaa0af04fab11ae01eb5f9fdb5f9c5b875506e210e71c07873528baa7f" +dependencies = [ + "tinyvec_macros", +] + +[[package]] +name = "tinyvec_macros" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" + +[[package]] +name = "tls_codec" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0de2e01245e2bb89d6f05801c564fa27624dbd7b1846859876c7dad82e90bf6b" +dependencies = [ + "tls_codec_derive", + "zeroize", +] + +[[package]] +name = "tls_codec_derive" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2d2e76690929402faae40aebdda620a2c0e25dd6d3b9afe48867dfd95991f4bd" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", +] + [[package]] name = "tokio" version = "1.52.1" @@ -3888,7 +4896,7 @@ checksum = "385a6cb71ab9ab790c5fe8d67f1645e6c450a7ce006a33de03daa956cf70a496" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -3930,6 +4938,7 @@ checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098" dependencies = [ "bytes", "futures-core", + "futures-io", "futures-sink", "futures-util", "pin-project-lite", @@ -3983,7 +4992,7 @@ checksum = "fec7c61a0695dc1887c1b53952990f3ad2e3a31453e1f49f10e75424943a93ec" dependencies = [ "async-trait", "axum", - "base64", + "base64 0.22.1", "bytes", "h2 0.4.13", "http 1.4.0", @@ -4065,7 +5074,7 @@ checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -4137,6 +5146,12 @@ version = "0.2.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ebc1c04c71510c7f702b52b7c350734c9ff1295c464a03335b00bb84fc54f853" +[[package]] +name = "untrusted" +version = "0.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a156c684c91ea7d62626509bce3cb4e1d9ed5c4d978f7b4352658f96a4c26b4a" + [[package]] name = "untrusted" version = "0.9.0" @@ -4149,7 +5164,7 @@ version = "3.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "dea7109cdcd5864d4eeb1b58a1648dc9bf520360d7af16ec26d0a9354bafcfc0" dependencies = [ - "base64", + "base64 0.22.1", "log", "percent-encoding", "rustls 0.23.40", @@ -4164,7 +5179,7 @@ version = "0.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e994ba84b0bd1b1b0cf92878b7ef898a5c1760108fe7b6010327e274917a808c" dependencies = [ - "base64", + "base64 0.22.1", "http 1.4.0", "httparse", "log", @@ -4213,7 +5228,9 @@ version = "1.23.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ddd74a9687298c6858e9b88ec8935ec45d22e8fd5e6394fa1bd4e99a87789c76" dependencies = [ + "getrandom 0.4.2", "js-sys", + "serde_core", "wasm-bindgen", ] @@ -4223,6 +5240,12 @@ version = "0.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ba73ea9cf16a25df0c8caa16c51acb937d5712a8429db78a3ee29d5dcacd3a65" +[[package]] +name = "vcpkg" +version = "0.2.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" + [[package]] name = "version_check" version = "0.9.5" @@ -4310,7 +5333,7 @@ dependencies = [ "bumpalo", "proc-macro2", "quote", - "syn", + "syn 2.0.117", "wasm-bindgen-shared", ] @@ -4519,7 +5542,7 @@ version = "44.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b2004f7c86ebeb116550655377cdf16dbf7b03ae5aa6b4b1c1458cfa23aaa306" dependencies = [ - "base64", + "base64 0.22.1", "directories-next", "log", "postcard", @@ -4542,7 +5565,7 @@ dependencies = [ "anyhow", "proc-macro2", "quote", - "syn", + "syn 2.0.117", "wasmtime-internal-component-util", "wasmtime-internal-wit-bindgen", "wit-parser 0.246.2", @@ -4653,7 +5676,7 @@ checksum = "d6af582ec18b674bf7a17775d6fbfbddfcc143f0edbd89c9c1778239c8aa92ed" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -4716,6 +5739,26 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "wasmtime-wasi-http" +version = "44.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5798e6e48005406983ba6a7b27f3832f1571e53f1401ef2d60111c96eab534b4" +dependencies = [ + "async-trait", + "bytes", + "futures", + "http 1.4.0", + "http-body 1.0.1", + "http-body-util", + "hyper 1.9.0", + "tokio", + "tracing", + "wasmtime", + "wasmtime-wasi", + "wasmtime-wasi-io", +] + [[package]] name = "wasmtime-wasi-io" version = "44.0.0" @@ -4780,6 +5823,92 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "webauthn-attestation-ca" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6475c0bbd1a3f04afaa3e98880408c5be61680c5e6bd3c6f8c250990d5d3e18e" +dependencies = [ + "base64urlsafedata", + "openssl", + "openssl-sys", + "serde", + "tracing", + "uuid", +] + +[[package]] +name = "webauthn-rs" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6c548915e0e92ee946bbf2aecf01ea21bef53d974b0793cc6732ba81a03fc422" +dependencies = [ + "base64urlsafedata", + "serde", + "tracing", + "url", + "uuid", + "webauthn-rs-core", +] + +[[package]] +name = "webauthn-rs-core" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "296d2d501feb715d80b8e186fb88bab1073bca17f460303a1013d17b673bea6a" +dependencies = [ + "base64 0.21.7", + "base64urlsafedata", + "der-parser 9.0.0", + "hex", + "nom", + "openssl", + "openssl-sys", + "rand 0.9.4", + "rand_chacha 0.9.0", + "serde", + "serde_cbor_2", + "serde_json", + "thiserror 1.0.69", + "tracing", + "url", + "uuid", + "webauthn-attestation-ca", + "webauthn-rs-proto", + "x509-parser 0.16.0", +] + +[[package]] +name = "webauthn-rs-proto" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c37393beac9c1ed1ca6dbb30b1e01783fb316ab3a45d90ecd48c99052dd7ef1e" +dependencies = [ + "base64 0.21.7", + "base64urlsafedata", + "serde", + "serde_json", + "url", +] + +[[package]] +name = "webpki-roots" +version = "0.26.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "521bc38abb08001b01866da9f51eb7c5d647a19260e00054a8c7fd5f9e57f7a9" +dependencies = [ + "webpki-roots 1.0.9", +] + +[[package]] +name = "webpki-roots" +version = "1.0.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7dcd9d09a39985f5344844e66b0c530a33843579125f23e21e9f0f220850f22a" +dependencies = [ + "rustls-pki-types", +] + [[package]] name = "wiggle" version = "44.0.0" @@ -4803,7 +5932,7 @@ dependencies = [ "heck", "proc-macro2", "quote", - "syn", + "syn 2.0.117", "wasmtime-environ", "witx", ] @@ -4816,7 +5945,7 @@ checksum = "aa7c29fcf738630cba4e35f1805da5e42dde20ee9809ee9202b0648ae671602f" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", "wiggle-generate", ] @@ -4891,7 +6020,7 @@ checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -4902,7 +6031,7 @@ checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -5078,7 +6207,7 @@ dependencies = [ "heck", "indexmap 2.14.0", "prettyplease", - "syn", + "syn 2.0.117", "wasm-metadata", "wit-bindgen-core", "wit-component", @@ -5094,7 +6223,7 @@ dependencies = [ "prettyplease", "proc-macro2", "quote", - "syn", + "syn 2.0.117", "wit-bindgen-core", "wit-bindgen-rust", ] @@ -5182,6 +6311,53 @@ dependencies = [ "tap", ] +[[package]] +name = "x509-cert" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1301e935010a701ae5f8655edc0ad17c44bad3ac5ce8c39185f75453b720ae94" +dependencies = [ + "const-oid 0.9.6", + "der", + "spki", + "tls_codec", +] + +[[package]] +name = "x509-parser" +version = "0.16.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fcbc162f30700d6f3f82a24bf7cc62ffe7caea42c0b2cba8bf7f3ae50cf51f69" +dependencies = [ + "asn1-rs 0.6.2", + "data-encoding", + "der-parser 9.0.0", + "lazy_static", + "nom", + "oid-registry 0.7.1", + "rusticata-macros", + "thiserror 1.0.69", + "time", +] + +[[package]] +name = "x509-parser" +version = "0.18.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d43b0f71ce057da06bc0851b23ee24f3f86190b07203dd8f567d0b706a185202" +dependencies = [ + "asn1-rs 0.7.2", + "aws-lc-rs", + "data-encoding", + "der-parser 10.0.0", + "lazy_static", + "nom", + "oid-registry 0.8.1", + "rusticata-macros", + "thiserror 2.0.18", + "time", +] + [[package]] name = "xattr" version = "1.6.1" @@ -5198,6 +6374,31 @@ version = "0.13.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "66fee0b777b0f5ac1c69bb06d361268faafa61cd4682ae064a171c16c433e9e4" +[[package]] +name = "yansi" +version = "1.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cfe53a6657fd280eaa890a3bc59152892ffa3e30101319d168b781ed6529b049" + +[[package]] +name = "yasna" +version = "0.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e17bb3549cc1321ae1296b9cdc2698e2b6cb1992adfa19a8c72e5b7a738f44cd" +dependencies = [ + "time", +] + +[[package]] +name = "yasna" +version = "0.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b5f6765e852b9b4dc8e2a76843e4d64d1cea8e79bcde0b6901aea8e7c7f08282" +dependencies = [ + "bit-vec", + "time", +] + [[package]] name = "yoke" version = "0.8.2" @@ -5217,7 +6418,7 @@ checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", "synstructure", ] @@ -5238,7 +6439,7 @@ checksum = "70e3cd084b1788766f53af483dd21f93881ff30d7320490ec3ef7526d203bad4" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] @@ -5258,7 +6459,7 @@ checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", "synstructure", ] @@ -5267,6 +6468,20 @@ name = "zeroize" version = "1.8.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b97154e67e32c85465826e8bcc1c59429aaaf107c1e4a9e53c8d8ccd5eff88d0" +dependencies = [ + "zeroize_derive", +] + +[[package]] +name = "zeroize_derive" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3c50655cbb0fe3fc43170059e702f1ce5e19b84cec58dc87b037a09935c2f328" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", +] [[package]] name = "zerotrie" @@ -5298,7 +6513,7 @@ checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.117", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index b2e81a9..9ae2765 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,15 +1,16 @@ [workspace] resolver = "2" members = [ + "crates/nitro-attestation", + "crates/nitro-nsm", "crates/s3fs-core", - "crates/s3fs-wasmtime", - "crates/s3fs-runner", + "runtime", ] [workspace.package] version = "0.1.0" edition = "2021" -rust-version = "1.78" +rust-version = "1.88" license = "Apache-2.0" repository = "https://github.com/joshua/s3-wasi-fs" @@ -28,6 +29,16 @@ hashlink = "0.9" bitvec = "1" percent-encoding = "2" +# Scrubs key material on drop. Enclave memory is unreadable from outside, but +# key bytes should not outlive their use inside it either. +zeroize = { version = "1", features = ["zeroize_derive"] } +# Merkle-tree block checksums and the root-record hash chain. BLAKE3 is the +# hot path (every block read verifies one), so its SIMD backends matter. +blake3 = "1" +# AEAD (AES-256-GCM), Ed25519 root signatures, HKDF key derivation. aws-lc-rs +# is the AWS-native, FIPS-capable stack — the same primitives KMS speaks. +aws-lc-rs = { version = "1", default-features = false, features = ["aws-lc-sys"] } + aws-config = { version = "1", default-features = false } aws-sdk-s3 = { version = "1", default-features = false } aws-smithy-runtime-api = "1" diff --git a/README.md b/README.md index c03c35a..c57b07d 100644 --- a/README.md +++ b/README.md @@ -1,169 +1,641 @@ -# s3-wasi-fs +# enclave-runtime -A [WASI Preview 2](https://github.com/WebAssembly/WASI) `wasi:filesystem@0.2.x` implementation backed by S3. +**Stateful WebAssembly services inside AWS Nitro Enclaves—with encrypted storage, attested connections, passkey authorization, and work that continues after the client disconnects.** -A Wasm component running inside Wasmtime sees a normal POSIX filesystem; the bytes land in an S3 bucket. Designed for [AWS Nitro Enclaves](https://aws.amazon.com/ec2/nitro/nitro-enclaves/) where a guest needs durable storage but the enclave can't mount FUSE — but works anywhere Wasmtime runs. +Write an application as a WASI component. Give it ordinary files, a SQLite database, HTTP handlers, and optional background callbacks. The runtime provides the machinery around it: an S3-backed filesystem, HTTPS terminated inside the enclave, tenant isolation, an attestation-aware authentication flow, durable schedules, outbound connections, and device wake notifications. -## Status +The project began as **s3-wasi-fs**. The filesystem is now the foundation of a larger runtime, and remains independently usable through `s3fs-core` without Wasmtime or enclave hardware. -Pre-1.0. Three workspace crates plus an example guest: +A client can check **which runtime booted, which application it loaded, and which TLS certificate belongs to that enclave** before authorizing an interaction. The application can persist state without exposing its filenames or plaintext blocks to the storage operator. That combination is useful for private application backends, personal agents, and interactive services that must keep working while their owner's phone is offline. -| Crate | Purpose | +> **Pre-1.0, under active correctness review.** The implementation includes the storage engine, serving runtime, KMS recipient flow, and deployment tooling. It is not a production-readiness claim. Real Nitro/KMS validation and garbage collection remain outstanding. + +## Contents + +- [What has been built](#what-has-been-built) +- [Try it locally](#try-it-locally) +- [Architecture and trust boundaries](#architecture-and-trust-boundaries) +- [The encrypted filesystem](#the-encrypted-filesystem) +- [Boot, state identity, and key release](#boot-state-identity-and-key-release) +- [Serving and verifying an application](#serving-and-verifying-an-application) +- [Passkeys, tenants, and interaction approval](#passkeys-tenants-and-interaction-approval) +- [Work beyond a request](#work-beyond-a-request) +- [Running SQLite](#running-sqlite) +- [Time, entropy, and guest output](#time-entropy-and-guest-output) +- [Configuration reference](#configuration-reference) +- [Build, test, and contribute](#build-test-and-contribute) +- [Build and deploy an enclave](#build-and-deploy-an-enclave) +- [Use the storage engine directly](#use-the-storage-engine-directly) +- [Limits and remaining work](#limits-and-remaining-work) +- [Repository guide](#repository-guide) + +## What has been built + +| Capability | Implementation and practical result | |---|---| -| [`s3fs-core`](crates/s3fs-core/) | Pure async S3-backed filesystem engine. `Backend` trait with in-memory and AWS S3 backends, inode tree with race-three lookup, buffer pool, MPU state machine, public `Fs` handle. **Zero wasmtime dependency** — usable from any host. | -| [`s3fs-wasmtime`](crates/s3fs-wasmtime/) | Wasmtime host bindings. Plugs into a `Linker` and exposes `wasi:filesystem` to a guest. Reuses `wasmtime-wasi` for everything else. | -| [`s3fs-runner`](crates/s3fs-runner/) | CLI binary that loads a `.wasm` component and runs it against a configured S3 bucket. | -| [`examples/guest-fsdemo`](examples/guest-fsdemo/) | Example Wasm component. Exercises mkdir / write / sync / read / rename / SQLite end-to-end. | +| **Encrypted, verifiable storage** | Copy-on-write block trees, AES-256-GCM encryption, BLAKE3 checksums, Ed25519-signed root records, and a hash-chained commit history. S3 contains opaque slabs rather than one object per pathname. | +| **A useful filesystem** | Reads, writes, sparse growth, directories, atomic rename, hard links, symlinks, timestamps, open-file unlink, and read-only historical snapshots. The guest uses `wasi:filesystem@0.2.x`. | +| **State-aware boot** | Explicit genesis, resume, upgrade, and refusal paths; an attested origin receipt ties a filesystem to its configured buckets, genesis root, and sealed-key reference. Retained reads look beneath S3 delete markers. | +| **Measured application loading** | The runtime fetches a component at boot, extends and locks PCR16, then asks for its key. PCR0 identifies the runtime image and measured configuration. A guest update need not rebuild the runtime image. | +| **KMS recipient key release** | `GenerateDataKey` and `Decrypt` use an enclave recipient; SSM holds the KMS ciphertext, and the roots bucket holds a checked pointer. The implementation and envelope tests exist; AWS hardware integration still needs validation. | +| **HTTPS and attestation** | Enclave-owned TLS keys, ACME TLS-ALPN-01, encrypted certificate caching, and nonce-bound attestation on `/auth/*` responses. Verification checks the actual connection certificate. | +| **Passkeys and tenant isolation** | WebAuthn registration, user verification, single-use interaction tokens, a filesystem scope per tenant, and a separate guest execution slot for each tenant. Native Android origins are configurable. | +| **HTTP/2 and bidirectional gRPC** | HTTP/1.1 and HTTP/2 share the listener. Guests can answer while a request is still arriving, with backpressure and trailers; a `tonic` client tests the example guest's framing. | +| **Durable background tasks** | Tenant-local schedules, recurring jobs, bounded retries, persisted status/results, and callbacks under the same tenant lock as interactive work. | +| **Connections held by the runtime** | Durable connection instructions, an SSE receive path, POST-based replies, reconnect backoff, and a fresh guest callback for incoming messages. A counterparty can initiate while the user is absent. | +| **Device wake notifications** | Runtime-owned Firebase Cloud Messaging integration, tenant-scoped device enrollment, coalescing, bounded queues, and content-free wake signals. | +| **Controlled guest egress** | Denied by default; a deployment may allow exact HTTP(S) origins. The production image measures this policy alongside the application environment. | +| **Operational tooling** | Structured guest logs, optional CloudWatch forwarding, PTP/NSM adapters, reproducible EIF builds, a QEMU development enclave, persistent development stores, and Packer/OpenTofu deployment definitions. | + +The examples make these capabilities tangible: [`guest-http`](examples/guest-http/) exercises state, isolation, tasks, and notifications; [`guest-grpc`](examples/guest-grpc/) demonstrates interactive streaming; [`guest-sqlite`](examples/guest-sqlite/) drives a real C database through the filesystem. The gRPC signing-session example exchanges demonstration messages; it does not implement a cryptographic signing protocol. + +## Try it locally + +There are two useful starting points. Use the host tests or [embed `s3fs-core`](#use-the-storage-engine-directly) to explore the storage engine without an enclave. Use the QEMU harness to develop against the complete measured boot and serving path. -**Test coverage:** 171 unit tests + 10 MinIO integration tests. End-to-end SQLite-on-S3 demo works. +### 1. Check the core on your machine -## Quick start: run a Wasm guest against MinIO +From the repository root, with Rust and its native build prerequisites installed: ```bash -# 1. Spin up MinIO -docker run -d --rm --name s3fs-mio -p 9000:9000 \ - -e MINIO_ROOT_USER=minioadmin -e MINIO_ROOT_PASSWORD=minioadmin \ - minio/minio server /data -aws --endpoint-url http://127.0.0.1:9000 \ - --region us-east-1 \ - s3api create-bucket --bucket demo - -# 2. Build the example guest (requires wasi-sdk for bundled SQLite) -curl -sLO https://github.com/WebAssembly/wasi-sdk/releases/download/wasi-sdk-25/wasi-sdk-25.0-x86_64-linux.tar.gz -mkdir -p ~/wasi-sdk && tar xf wasi-sdk-25.0-x86_64-linux.tar.gz -C ~/wasi-sdk --strip-components=1 -cd examples/guest-fsdemo -CC_wasm32_wasip2=$HOME/wasi-sdk/bin/clang \ -AR_wasm32_wasip2=$HOME/wasi-sdk/bin/ar \ -CFLAGS_wasm32_wasip2="--sysroot=$HOME/wasi-sdk/share/wasi-sysroot -DSQLITE_THREADSAFE=0 -DHAVE_USLEEP=1" \ - cargo build --release --target wasm32-wasip2 - -# 3. Build the runner and execute the guest -cd ../.. -cargo build --release -p s3fs-runner -./target/release/s3fs-runner \ - --bucket demo \ - --region us-east-1 \ - --endpoint http://127.0.0.1:9000 \ - --access-key-id minioadmin \ - --secret-access-key minioadmin \ - --force-path-style \ - --component examples/guest-fsdemo/target/wasm32-wasip2/release/guest-fsdemo.wasm -# → prints "OK" -``` - -The guest writes a real SQLite database into the bucket through `wasi:filesystem`. After the run, `aws s3 ls s3://demo/` shows the bucket cleaned up by the guest's own `remove_dir_all` calls. - -## Architecture - -``` -┌──────────────────────────────────────────────────────────────────┐ -│ wasm guest component │ -│ (Rust binary using std::fs / SQLite / ...) │ -└────────────────────────────────┬─────────────────────────────────┘ - │ wasi:filesystem@0.2.6 - ▼ -┌──────────────────────────────────────────────────────────────────┐ -│ wasmtime + wasmtime-wasi (io / cli / clocks / random / sockets) │ -│ + s3fs-wasmtime (filesystem only) │ -│ Descriptor / DirectoryEntryStream resources │ -│ S3InputStream / S3OutputStream over wasi:io │ -└────────────────────────────────┬─────────────────────────────────┘ - │ Fs handle API - ▼ -┌──────────────────────────────────────────────────────────────────┐ -│ s3fs-core::Fs │ -│ openat path resolution + symlink follow (depth ≤ 40) │ -│ inode tree (race-three lookup, snapshot listings) │ -│ buffer pool (memory-bounded LRU, dirty pinning) │ -│ MPU state machine (lazy begin, copy_unmodified_parts, │ -│ parallel UploadPart/UploadPartCopy) │ -└────────────────────────────────┬─────────────────────────────────┘ - │ Backend trait - ▼ - ┌─────────────┴─────────────┐ - ▼ ▼ - AwsS3Backend MemoryBackend - (aws-sdk-s3, S3-compatible) (in-memory, tests) -``` - -## Compatibility Matrix - -The full matrix — what's POSIX-equivalent, what's weakened, what's unsupported — lives at [`docs/COMPATIBILITY.md`](docs/COMPATIBILITY.md). Headlines: - -- ✅ **All read/write/stat/listdir/sync/mkdir/unlink/rmdir/rename ops** including recursive directory rename. -- ✅ **Symlinks** with cycle detection (depth 40), atomic create via `If-None-Match: *`. -- ✅ **`O_EXCL`** atomic create via S3 conditional PUT. -- ✅ **`O_TRUNC`** synchronous zero-byte PUT on open. -- ✅ **`set_size`** (truncate + capped grow) and **`set_times{,-at}`** (persisted as `x-amz-meta-s3wasifs-{atime,mtime}`). -- ✅ **In-place updates** of large files via MPU + `UploadPartCopy` for unchanged parts (GeeseFS-parity). -- ✅ **Eager background `UploadPart`** — `pwrite` of a fully-filled part hands off to a long-running flusher task; `sync` drains the parked replies. Caps in-flight uploads at `max_parallel_parts` across the whole `Fs`. -- ✅ **Async rename** — `Fs::rename` rewires the inode tree synchronously and returns immediately; the worker handles `CopyObject` + `DeleteObject` (or paginated recursion for directories) in the background. Reads/writes against the new path during the in-flight window resolve to the old key via `current_s3_key`. -- ⚠️ **`rename`** is not POSIX-atomic (CopyObject + DeleteObject is two calls); a host crash mid-window leaves both keys. -- ⚠️ **`unlink` while fd is open** doesn't keep the file readable on stale handles. -- ⚠️ **Concurrent writers to the same key**: last writer wins (no fencing). -- ❌ **`link_at` (hardlinks)** — S3 has no shared-identity-across-keys. - -## Building from source +cargo test -p s3fs-core --lib +cargo test --workspace --lib +``` + +These commands need neither Docker nor a live AWS account. The workspace library tests include cryptography, storage, authentication, attestation verification, and runtime policy tests. Tests marked ignored require additional fixtures; a default pass does not exercise them. + +### 2. Start the complete development enclave + +The harness targets Linux with working KVM and vsock, Docker, Nix with flakes, Rust, and the `wasm32-wasip2` target. It also uses `python3`, `jq`, `curl`, and `openssl`. Run these commands from the repository root: ```bash -git clone -cd s3-wasi-fs -cargo build --release --workspace --features aws # AWS feature builds AwsS3Backend -cargo test --workspace --features aws --lib # 171 unit tests -cargo clippy --workspace --features aws --all-targets -- -D warnings +rustup target add wasm32-wasip2 +sudo modprobe vsock_loopback + +docker build -t s3fs-qemu-nitro:latest deploy/qemu-nitro +cargo install vhost-device-vsock --version 0.3.0 --locked \ + --root target/qemu-nitro/tools + +# Builds the example guest and starts the supporting services. +deploy/qemu-nitro/dev-enclave.sh --keep-store ``` -To run the MinIO integration suite (requires Docker): +The first run builds an enclave image and can take substantially longer than a host test. The harness starts MinIO, the local ACME CA, networking, and the emulator; prints the URL, PCR0, PCR16, and trust-root path; then stays running. Its default URL is `https://127.0.0.1:8443`. + +**The emulator exercises the protocol, not AWS's hardware trust boundary.** Its master key is a development key. It generates its own attestation signing chain because QEMU's NSM cannot produce an AWS-signed document. Pebble supplies local ACME certificates. These are suitable for test data, including when this harness runs on a public host. + +### 3. Make an authenticated, stateful request + +In another terminal, substitute the two measurements printed by the harness: + +```bash +target/release/passkey-client \ + --url https://127.0.0.1:8443 \ + --state target/qemu-nitro/dev/alice.json \ + --trust-root target/qemu-nitro/dev/trust-root.der \ + --pcr0 --pcr16 \ + get --path /counter +``` + +Run it again with the same state file to observe the persistent counter. The client registers a software passkey for testing, verifies the enclave, and approves the request. A different state file registers a different tenant. The software authenticator is available through the `testing` feature and is not part of the production image. + +With `--keep-store`, stopping and restarting the harness preserves the filesystem and registered credentials. Read the new pins at each boot: the emulator's attestation root changes, and a different guest changes PCR16. `--fresh --keep-store` intentionally discards the retained development store. + +### 4. Bring your own component ```bash -cargo test -p s3fs-core --features aws --test minio_integration -- --ignored --test-threads=1 -# 10 tests, ~30s each (MinIO container per test) +deploy/qemu-nitro/dev-enclave.sh \ + --guest path/to/component.wasm \ + --keep-store \ + --guest-egress http://192.168.127.254:7070 \ + --guest-env SERVICE_URL=http://192.168.127.254:7070 +``` + +The guest must implement `wasi:http/incoming-handler`; optional tasks, notifications, and held connections use the [WIT interfaces](wit/). The default emulator image enables background tasks and therefore also requires `run-task`. For an HTTP-only component, add `--guest-env S3FS_BACKGROUND_TASKS=false`; the harness merges this into the measured runtime image environment, and the runtime filters it out of the guest environment. `192.168.127.254` reaches the development host through gvproxy. Omit the egress option when the guest needs no external service. + +The [development guide](docs/DEV_ENCLAVE.md) covers native app origins, public-domain ACME, real FCM, memory sizing, preserved stores, `--pack`/`--prebuilt`, and troubleshooting. The harness defaults to the HTTP example if `--guest` is omitted. + +**Why there is no bare-runtime laptop command here:** the serving binary measures its guest into PCR16 before booting the store. That requires an NSM; `--tls off` and a static master key do not bypass it. Ordinary host integration tests supply a fake NSM through the library, while QEMU supplies an emulated device. + +## Architecture and trust boundaries + +```mermaid +flowchart TB + Client[Native client and passkey] -->|HTTPS| Parent[Parent: gvproxy and vsock forwarding] + subgraph Enclave[Measured enclave image] + TLS[TLS and attested auth exchange] + Gate[WebAuthn gate and single-use tokens] + Runtime[Tenant locks, Wasmtime, background workers] + Guest[WASI component] + FS[Filesystem and copy-on-write block store] + NSM[NSM: entropy, PCRs, attestation] + TLS --> Gate --> Runtime --> Guest --> FS + NSM --> TLS + end + Parent -->|Encrypted traffic| TLS + FS -->|Encrypted slabs| Data[(S3 data bucket)] + FS -->|Signed roots and retained records| Roots[(S3 roots bucket)] + Runtime -->|Attested recipient| KMS[KMS and SSM] + Runtime -->|Configured capabilities| Services[Allowed origins, FCM, CloudWatch] ``` -## Embedding `s3fs-core` directly +The parent provides transport and decides whether the enclave runs. The design excludes it from plaintext application storage and enclave-terminated TLS when production key release and client verification are correctly configured. It can still stop the service, drop traffic, and observe traffic metadata. + +AWS Nitro attestation, KMS, S3's authenticated service responses, Object Lock enforcement, and the approved runtime/guest are part of the trust model. The storage operator's ability to alter objects is distinct from S3 itself lying about which versions exist. Root freshness on a cold mount also needs an external lower bound when newer history may be hidden. + +An approved guest is trusted with its tenant's data. Attestation identifies code; it does not prove that code is safe. Logging and allowed outbound destinations are deliberate disclosure channels and should be reviewed with the guest. + +## The encrypted filesystem + +### From a pathname to a root record + +```text +wasi:filesystem descriptors and streams + │ + ▼ +Fs: paths, links, directories, handles, buffered records + │ + ▼ +Directory B+tree ──► object IDs ──► dnode array + │ + indirect block trees + │ + encrypted blocks in slabs + │ + signed root record + chain +``` + +Files and directories are addressed by object ID. Directory entries live in a separator-indexed B+tree. Dnodes describe objects and point into variable-width indirect block trees. Modified paths are rebuilt copy-on-write, so a commit publishes a new tree while older roots continue to describe their original state. + +Blocks carry AES-256-GCM authentication and BLAKE3 checksums. Their authentication binds them to their expected position, preventing a valid block from simply being moved elsewhere in the tree. Keys derive from a 32-byte master secret and filesystem identity through HKDF. The data bucket holds slab objects; the roots bucket holds the signed history and runtime bootstrap objects. + +Verifying a root authenticates the pointers beneath it; each block is verified when read. Mounting does **not** eagerly read or scrub every file. + +### Commit and recovery + +A transaction stages changed blocks and dnodes, uploads its slabs, waits for those uploads, then publishes a signed root with a conditional PUT. Publishing that root is the visibility boundary. A failed upload must not publish a tree pointing to missing data. Failed file sync retains dirty buffers for a retry. + +The current transaction protocol also publishes a claim before a newly opened session encrypts blocks, and reclaims after an abandoned or failed transaction. Its purpose is to avoid reusing `(transaction group, block sequence)` as an AEAD nonce after a crash or competing mount. S3 accepts a conditional PUT over a delete marker, so publication then reads back the retained version: a mount whose record was not written first loses, even when its PUT succeeded. + +This is a **single-writer filesystem**. Independent writable mounts and multiple active schedulers are not a supported deployment model. A conflict poisons the losing mount rather than silently merging histories. + +### Filesystem behavior + +| Operation or property | Behavior | +|---|---| +| Read/write/stat/list/sync | Implemented through the core `Fs` API and WASI adapter. File writes are buffered; sync/close drives durability. | +| Rename | Directory-entry changes commit atomically; moving a directory does not copy every descendant. Moves into the directory's own subtree are rejected. | +| Sparse files | Growth creates holes; reads return zeroes without materializing all intervening blocks. | +| Hard links | Multiple directory entries reference one object, with a maintained link count. | +| Symlinks | Supported, with a default traversal limit of 40 and resolution confined to the guest's filesystem scope. | +| Unlink while open | The open handle retains access until its final close. | +| Identity and metadata | Object identity, link counts, and timestamps come from the filesystem rather than inferred S3 filenames. | +| Snapshots | Historical root records can be opened read-only through `Fs::open_snapshot` or the store API; they share unchanged blocks. They are not an automatic guest-visible snapshot directory. | +| Freshness | The session floor only rises. `--min-root-seq` supplies an external floor for a cold mount. | +| Space reclamation | Garbage collection is not implemented. Historical and orphaned slabs consume storage. | + +The detailed [compatibility matrix](docs/COMPATIBILITY.md) describes WASI/POSIX differences. This is a WASI filesystem implementation and embeddable Rust engine, not a host FUSE mount or a general replacement for a Linux filesystem. + +### Defaults and cost model + +| Store setting | Default | +|---|---:| +| Record size | 128 KiB | +| Maximum slab size | 64 MiB | +| Decrypted block-cache budget | 64 MiB | +| Concurrent slab PUTs | 8 | +| Root-chain links checked at mount | 1 | +| Per-root COMPLIANCE retention | 10 × 365 days | + +[`StoreConfig`](crates/s3fs-core/src/store/config.rs) validates these settings. Record sizes must be powers of two from 4 KiB through 1 MiB. Retention is finite: an operational retention policy must account for the history a deployment still relies on. + +S3 round trips dominate durable mutations. Batching application work reduces commit overhead; smaller writes may still rewrite a whole record and its tree path. Tenant handlers can execute concurrently, but their commits share one transaction lock. Snapshot sharing avoids copying the whole filesystem, while retaining old state still costs space for its changed blocks. + +## Boot, state identity, and key release + +### Explicit state origins + +A missing store is not automatically an invitation to create an empty replacement. The boot machine checks the retained state-origin receipt, the sealed-key object, the genesis root, and the record for the running runtime/guest pair. + +The origin commitment includes the filesystem ID, data and roots buckets, prefix, genesis root hash, and the SHA-256 of the sealed-key representation. It is encoded with a domain tag and CBOR, then hashed with BLAKE3. The receipt attests to that commitment. + +| Observed state | Outcome | +|---|---| +| No origin receipt and no existing state | Genesis: provision key material, create the store, publish origin and pair records. | +| State exists without its receipt, or receipt exists without required state | Refuse to boot. | +| Origin and this runtime/guest pair verify | Resume. | +| Origin verifies but this pair has no record | Upgrade path: record the new pair after key release and state verification. | +| Existing origin/pair record is inconsistent or invalid | Refuse to boot. | + +Bucket identity and policy live in the measured image. Otherwise, a host could point an approved runtime at a different empty bucket and make a new store look legitimate. + +[`Backend::get_retained_blob`](crates/s3fs-core/src/backend/mod.rs) reads object versions beneath delete markers. The S3 backend follows version-list pagination and fetches a selected version by ID. Missing version-list permissions cause a failure rather than a false report that no record exists. The retained version is the oldest one listed, and root publication uses the same read to confirm it wrote first. + +### KMS is implemented; hardware validation remains + +The production key source is [`KmsAttestedKey`](runtime/src/keys/kms.rs), selected explicitly with `--master-key-source kms`: + +1. Load the guest, extend PCR16 with its hash, and lock the register. +2. Generate an enclave recipient key and put its public key into an NSM attestation request. +3. At genesis, call KMS `GenerateDataKey` with the recipient; on resume, call `Decrypt` with the recipient. +4. Require and unwrap `CiphertextForRecipient` inside the enclave; the implementation does not use a plaintext response field as a fallback. +5. Keep the KMS ciphertext in SSM. Persist a pointer containing the parameter name, KMS key identity, encryption context, and ciphertext digest in the roots bucket. + +The encryption context includes filesystem identity and environment. [`recipient.rs`](runtime/src/keys/recipient.rs) parses and validates the CMS envelope. The origin receipt commits to the sealed pointer, and loading the SSM parameter checks the ciphertext digest. + +The deployment's KMS policy must constrain **both** `kms:GenerateDataKey` and `kms:Decrypt` using **both** `kms:RecipientAttestation:PCR0` and `kms:RecipientAttestation:PCR16`. The runtime implements the exchange; KMS policy is what enforces release. A broader alternate grant can undermine that policy. + +`--master-key-source static --master-key <64-hex-characters>` is the development alternative. It stores an explicitly unsealed representation and does not protect the key from its operator. Supplying a plaintext master key in KMS mode is refused. There is no implicit default key source. + +### Guest upgrades + +The runtime image names a guest object in the roots bucket. It measures the bytes it actually fetched: + +```text +fetch component → SHA-256(component) → extend and lock PCR16 → key release → mount +``` + +PCR16 is the SHA-384 extension of the zero register with that SHA-256 digest; it is not just the component's raw SHA-256. `nitro-attest --measure` and the guest-release build compute the value clients and policies must pin. + +A guest-only upgrade changes the object, approved PCR16, and client pins. Changes to runtime code or measured deployment configuration change PCR0 and require an image rebuild. Preserve compatibility for persisted application data and queued task payloads: outstanding work runs the newly approved guest. Origin and upgrade records describe provenance; they do not turn attestation into an application migration system. + +## Serving and verifying an application + +### HTTP, TLS, and streaming + +The guest implements `wasi:http/incoming-handler` and receives parsed HTTP requests. Rustls terminates TLS inside the enclave; the parent forwards encrypted traffic over gvproxy/vsock. HTTP/2 is negotiated through ALPN alongside HTTP/1.1. A gRPC guest can read request messages while writing its response, including response trailers. + +Production TLS uses ACME TLS-ALPN-01 on the forwarded TLS port. `--tls acme` requires domains and an ACME directory reachable from the enclave; `--acme-directory` can select a private CA. The QEMU harness uses Pebble by default. `--tls off` is a plaintext development mode, not an enclave confidentiality boundary. + +The runtime creates and retains the TLS private key inside its boundary. ACME account/certificate material is encrypted under a derived master key and cached outside the guest's preopen. This avoids exposing the TLS key as an application file and avoids reissuing a certificate on every restart. + +### Attest the connection before approving it + +| Header | Contract | +|---|---| +| `x-enclave-nonce` | Required on every request. Fresh client-chosen bytes, base64url without padding, 8–64 bytes after decoding. Missing/malformed values are rejected before routing. | +| `x-enclave-attestation` | Present on attested `/auth/*` responses; contains the base64-encoded COSE document. Guest responses have this header removed. | + +The document binds the client's nonce, the TLS leaf certificate actually used for that connection, and the component hash. Its `user_data` is: + +```text +0x12 0x20 || SHA256(TLS leaf DER) || 0x12 0x20 || SHA256(component) +``` + +A native client verifies the signature and certificate chain against its trusted Nitro root (or the explicitly pinned emulator root), document validity/freshness, its nonce, PCR0, PCR16, and the certificate hash. It then pins that certificate for the operation's TLS connection **before sending its token or sensitive payload**. Both measurements matter: an arbitrary runtime could write a convincing PCR16, while PCR0 alone no longer identifies the separately loaded guest. + +The normal challenge exchange performs attestation; a separate probe is unnecessary. `GET /auth/` can also obtain an attested method refusal for diagnostics. Guest responses do not incur another NSM signature. A client must require the proof on the authentication responses where it belongs and must not treat its intentional absence on guest responses as a downgrade. + +The document proves a connection and code identity. It does **not** sign a response body, prove a guest's result correct, or attest the latest filesystem root for every request. Ordinary browser JavaScript cannot inspect the TLS peer certificate needed by this flow; the supplied reference client and integration guide target native clients. + +The runtime checks the attestation header against a 16 KiB ceiling at startup. Client transports must budget for the whole response head, including the certificate chain; the existing client guidance uses at least 32 KiB. Actual NSM latency and sustainable authentication throughput remain hardware measurements to obtain. + +See [client integration](docs/CLIENT_INTEGRATION.md), [streaming](docs/STREAMING.md), and the standalone [`nitro-attest` verifier](crates/nitro-attestation/src/bin/nitro-attest.rs). + +## Passkeys, tenants, and interaction approval + +### One approval, one interaction + +When WebAuthn is configured, a request reaches the guest only after redeeming an unspent interaction token: + +```text +POST /auth/request/options {credential_id, method, path, query} + │ + ├─ verify attestation and pin the enclave's TLS certificate + ├─ obtain a user-verified passkey assertion + ▼ +POST /auth/request/verify {challenge_id, assertion} + │ + └─ receive a short-lived, single-use token + ▼ +application request Authorization: Bearer +``` + +The WebAuthn verifier checks the challenge, exact allowed origin, relying party, signature, and user verification. The runtime enforces expiry, one-time use, registered credentials, and revocation. Tokens contain 32 random bytes, are indexed by their SHA-256, and disappear at restart. Redemption consumes the token before dispatch; an invalid attempt does not leave it reusable. + +**The approval binds method, path, and query—not the request body or each stream message.** It authorizes an interaction, which may be an ordinary request or a bidirectional stream. Applications needing approval of exact transaction bytes need an additional application protocol. The gRPC demonstration should not be treated as transaction-signing authorization. + +Registration is open and creates a new empty tenant. It does not grant access to an existing tenant. This also means registration is not an admission-control or anti-abuse system. `--max-tenants` limits cached execution slots, not the total durable user population or all active resource use. + +Android app origins can be explicitly allowed as `android:apk-key-hash:`. The app/domain association and exact signing-certificate origin must match the deployment; adding an allowed origin changes measured image configuration. See the [client guide](docs/CLIENT_INTEGRATION.md) for the complete registration and approval exchange. + +### Tenant storage and instance lifetime + +Registration mints a random **16-byte** tenant ID and stores it with the credential. The runtime supplies tenant identity to the guest; a caller cannot select another tenant using a request header. A tenant sees its own directory under `/tenants/` as `/`. + +Absolute paths, `..`, absolute symlinks, and parent-descriptor traversal stay within that scope. The filesystem and decrypted-block cache are shared; each tenant has a separate lock and guest instance. Runtime records under `/runtime` remain outside the tenant preopen. + +A healthy HTTP instance may be reused **for the same tenant**. It is recycled after configured limits and discarded after failures or unsettled resources. Background and message callbacks use fresh instances under the tenant lock. Never rely on warm memory for durable state: eviction, traps, restarts, and background activity can remove it. Files a discarded instance left open are released without being flushed; close or sync what must persist before the call returns. + +One tenant's interactive requests, background tasks, and incoming-message callbacks serialize with each other. Different tenants can execute concurrently; commits still serialize through the shared store. An open inbound stream holds its tenant's slot through the invocation/body lifecycle, so the same tenant's next operation may wait. Anonymous calls, where authentication is disabled, share one execution slot over the runtime filesystem. -The engine is usable without the wasmtime layer. Useful for FUSE adapters, gateways, or custom hosts. +## Work beyond a request + +### Durable background tasks + +Enable `S3FS_BACKGROUND_TASKS=true` with authentication and a guest exporting `enclave:tasks/background@0.1.0`'s `run-task`. The [`queue` interface](wit/tasks/tasks.wit) supports enqueue, status, cancel, and forget. Tenant identity comes from the invocation, not a guest-supplied argument. + +An authenticated interactive call can schedule work for later or specify a recurring interval. Records live in the encrypted filesystem and survive restart. Workers honor per-tenant serialization, defer busy tenants without spending retries, enforce an execution deadline, and store bounded errors and results. Recurring intervals must be at least one second; payload and result limits are 64 KiB. + +Execution may repeat after a crash. Handlers must deduplicate using the supplied **run ID**; a recurring task's occurrences have different run IDs. Writing an application file and enqueueing a task are separate mutations, not one atomic application transaction. Revoking a passkey does not automatically cancel schedules it authorized. Background callbacks cannot grant themselves more queue authority. + +The [background-task guide](docs/BACKGROUND_TASKS.md) describes retries, occurrence identity, cancellation, limits, and upgrade behavior. The scheduler requires one active owning enclave; distributed leases and fencing are not implemented. + +### Connections a counterparty can use first + +The [`enclave:streams/connection@0.1.0` interface](wit/stream/stream.wit) lets an interactive guest ask the runtime to maintain a connection to an allowed origin: + +```text +stream-open(id, origin) maintain a durable connection instruction +stream-close(id) remove that instruction +stream-send(id, payload) send through the runtime +stream-status(id) inspect connection status +on-message(id, message-id, payload) → reply bytes +``` + +The runtime keeps the network connection while no guest is executing. Incoming SSE events invoke `on-message` in a fresh, tenant-scoped instance. Nonempty replies are POSTed back; a guest may also send from another active invocation. The peer contract is: + +```text +GET /escrow/stream?id=- +POST /escrow/send?id=- +``` + +SSE `data` contains base64 bytes; an `id` supplies message identity, with a content-derived fallback when absent. The tenant prefix prevents the same guest-local name from colliding across users. The runtime limits a tenant to eight connections and each message to 256 KiB, with reconnect delays from one second up to five minutes. The origin policy is checked again on reconnect. + +Connection instructions are durable; messages are not a durable inbox/outbox. Deduplicate replayed IDs in the application and arrange replay/acknowledgment with the peer. Do not infer exactly-once delivery or lossless recovery from an SSE reconnect. Closing and reopening a stream, to the same origin or another, drops the old connection and dials a new one. + +### Wake a device without sending its private content + +The [`enclave:notify/notify@0.1.0`](wit/notify/notify.wit) capability lets a tenant enroll devices and ask the runtime to send a data-only FCM wake. The app fetches the details through an authenticated connection after waking. + +Wake payloads contain a schema version, category, and optional tenant-local reference. There is no user-facing title/body or notification object. Choose opaque labels when the category or reference could reveal sensitive meaning; the provider sees the destination, labels, and timing. + +Device enrollment/removal requires an interactive invocation. Background work can wake already enrolled devices, but cannot enroll destinations or read stored device tokens back. The runtime owns OAuth, the service-account credential, HTTPS, coalescing, backoff, and a bounded queue. Delivery is best effort and is not a durable audit or messaging channel. + +Configure `S3FS_FCM_PROJECT_ID` plus one credential source: literal service-account JSON for development, or an SSM parameter name for deployment. The local harness supplies an FCM stub. See [notifications](docs/NOTIFICATIONS.md) for the WIT contract, limits, and operational behavior. + +### Guest egress + +The default policy refuses outgoing HTTP. `--guest-egress-origin` / `S3FS_GUEST_EGRESS_ORIGINS` permits exact scheme/host/port origins. HTTPS uses the runtime's configured verification path and compiled roots; direct guest socket access is not granted. The allowlist also governs held connections. + +This permits useful integrations without granting arbitrary network access, but an allowed destination can receive anything the approved guest sends to it. Production origin lists and guest environment values are baked into the image and affect PCR0. + +## Running SQLite + +SQLite is the main application-level filesystem workload: real C code performs page-granular random I/O, rollback-journal updates, and integrity checks. [`guest-sqlite`](examples/guest-sqlite/) covers DDL, transactions, savepoints, constraints, joins, CTEs, window functions, blobs, triggers, schema changes, `VACUUM`, JSON, FTS5, and R-Tree where available. + +Build it using wasi-sdk: + +```bash +scripts/wasi-sdk.sh +scripts/build-guest.sh sqlite + +deploy/qemu-nitro/dev-enclave.sh \ + --guest examples/guest-sqlite/target/wasm32-wasip2/release/guest-sqlite.wasm \ + --guest-env S3FS_BACKGROUND_TASKS=false +``` + +SQLite exports the HTTP handler, not `run-task`, so the override disables the emulator's default background-task startup requirement. Using the printed pins, request `/` with `passkey-client` to run the workload. The response reports `OK` or a failure; timings go to guest output. Each run removes and recreates its benchmark database, so use a dedicated development tenant and do not treat `/` as a health check. `SQLITE_SCALE` controls the workload size; the current example defaults to 2,000 rows. + +The two required pragmas are: + +```rust +conn.pragma_update(None, "temp_store", "MEMORY")?; +conn.pragma_update(None, "locking_mode", "EXCLUSIVE")?; +``` + +SQLite's temporary-directory probing and advisory locking assume syscalls that this WASI environment does not supply. Use the default rollback-journal mode (`DELETE`). WAL requires the shared-memory/mapping facilities this configuration lacks. Runtime serialization protects one tenant from concurrent callbacks; the guest must still avoid opening competing SQLite connections inside its own invocation. + +### Recorded workload results + +The following are the repository's earlier measurements against local MinIO with 20,000 accounts and 40,000 entries. They were **not rerun for this documentation update**, are not current release benchmarks, and should not be extrapolated to remote S3 latency. + +| Phase | Elapsed | Rate | +|---|---:|---:| +| Bulk insert, 20,000 rows in one transaction | 142 ms | 140,884 rows/s | +| Point select by row ID | 2.5 ms | 405,948 queries/s | +| Indexed select | 1.3 ms | 785,850 queries/s | +| Update 500 rows in one transaction | 63 ms | 7,963 rows/s | +| Blob write | 128 ms | 18.5 MB/s | +| Blob read and verify | 41 ms | 57.8 MB/s | +| Incremental blob I/O, 512 KiB scattered | 242 ms | 2.2 MB/s | +| `VACUUM` | 471 ms | — | +| `PRAGMA integrity_check` | 136 ms | — | +| 25 inserts, autocommit | 1,796 ms | 14 rows/s | +| 25 inserts, one SQL transaction | 69 ms | 364 rows/s | + +The practical lesson is to batch writes. A SQL transaction is not necessarily one block-store commit—journal and filesystem sync operations matter—but it can greatly reduce durable round trips compared with repeated autocommit statements. + +## Time, entropy, and guest output + +### Clock and randomness + +The runtime supplies guest clocks and random sources through its WASI environment. Filesystem timestamp operations requesting “now” use the same wall-clock adapter the guest sees. + +| Setting | Modes | +|---|---| +| `--clock-source` | `ptp` requires the configured PTP device; `host` uses the system clock; `auto` tries PTP and warns when falling back. | +| `--ptp-device` | PTP device path, default `/dev/ptp0`. Availability must be checked in the target environment. | +| `--random-source` | `nsm` requires the NSM entropy source; `host` uses host randomness; `auto` selects an available source. | +| `--nsm-device` | NSM device path, default `/dev/nsm`. Host randomness does not supply PCRs or attestation. | + +Production configuration explicitly requests trusted devices. Automatic fallbacks are convenient for host tests but should not silently define a deployment's trust policy. The diagnostic path and [`deploy/ptp-check.sh`](deploy/ptp-check.sh) help inspect clock availability; [`run-selftest.sh`](deploy/qemu-nitro/run-selftest.sh) exercises NSM behavior in the emulator. + +### Logs are structured output, not an audit proof + +Guest `stdout` and `stderr` pass through bounded line framing and a nonblocking logging queue. They become structured events tagged with their stream; guest text is an escaped field value rather than log syntax. Long lines and dropped records are accounted for. A guest cannot block its filesystem or request indefinitely by waiting on a remote log collector. + +The default sink is tracing. Setting `S3FS_GUEST_LOG_GROUP` and `S3FS_GUEST_LOG_STREAM` enables CloudWatch forwarding to pre-created resources. The forwarder batches JSON events, retries transient failures with backoff, bounds retained work, and counts drops. Queued output can be lost on abrupt shutdown. + +Guest output can contain secrets the guest chooses to print. Console logs are visible to the host, and a CloudWatch reader can read forwarded content. The parent role can also write forged records to the same CloudWatch stream. These are operational logs, not tamper-proof evidence. IMDSv2 credential resolution through the deployed networking path and real CloudWatch delivery still need hardware validation. + +## Configuration reference + +The binary's [`Cli`](runtime/src/main.rs) is the definitive flag reference: + +```bash +cargo run -p enclave-runtime --bin enclave-runtime -- --help +``` + +Configuration embedded by Nix is part of the measured image. Changing a bucket, relying party, allowed origin, logging destination, or guest environment changes PCR0. The `S3FS_` environment prefix is retained from the filesystem's original name. + +| Area | Main flags / environment | +|---|---| +| Store identity | `--bucket`, `--roots-bucket`, `--bucket-prefix`, `--fs-id`; `S3FS_BUCKET`, `S3FS_ROOTS_BUCKET`, `S3FS_BUCKET_PREFIX`, `S3FS_ID` | +| S3 access | `--region`, `--endpoint`, `--force-path-style`; AWS credentials/default chain; optional session token | +| Freshness | `--min-root-seq` / `S3FS_MIN_ROOT_SEQ` | +| Key source | Required `--master-key-source kms\|static` / `S3FS_MASTER_KEY_SOURCE` | +| KMS/SSM | `--kms-key-id`, `--master-key-parameter`, `--environment`; separate KMS/SSM endpoint overrides | +| Development key | `--master-key` / `S3FS_MASTER_KEY`, only with the static source | +| Component | `--guest-object` / `S3FS_GUEST_OBJECT`, or `--guest-path` / `S3FS_GUEST_PATH` | +| Listener and TLS | `--http-listen`, `--tls`, repeatable `--tls-domain`, `--acme-contact`, `--acme-directory` | +| WebAuthn | `--webauthn-rp-id`, `--webauthn-origin`, repeatable `--webauthn-allowed-origin` | +| Guest network | Repeatable `--guest-egress-origin` / `S3FS_GUEST_EGRESS_ORIGINS` | +| Guest environment | `--no-inherit-env`, repeatable `--guest-env NAME=VALUE` or `--guest-env NAME` | +| Background work | `--background-tasks true`, concurrency, timeout, total-record and per-tenant limits | +| Notifications | `S3FS_FCM_PROJECT_ID` plus `S3FS_FCM_SERVICE_ACCOUNT` or `S3FS_FCM_SERVICE_ACCOUNT_PARAMETER` | +| Logging | `S3FS_GUEST_LOG_GROUP`, `S3FS_GUEST_LOG_STREAM`; tracing filtering through `RUST_LOG` | + +By default the guest inherits the runtime environment after `AWS_*` and `S3FS_*` filtering. That is appropriate only when the environment is deliberately curated. Host-side callers should use `--no-inherit-env` and explicitly name values; the denylist does not know every application's secret variable. + +| Runtime limit | Default | +|---|---:| +| Request progress timeout | 30 s | +| S3 operation timeout | 30 s | +| WebAuthn challenge lifetime | 60 s | +| Unspent interaction token lifetime | 60 s | +| Maximum interaction lifetime | 300 s | +| Cached tenant limit | 64 | +| Tenant idle timeout | 900 s | +| Requests before instance recycling | 10,000 | +| Background concurrency | 1 | +| Background attempt timeout | 30 s | +| Background record limit, global / per tenant | 1,024 / 64 | + +The progress watchdog and interaction lifetime solve different problems: an active stream can outlive a short request timeout, while the overall interaction remains bounded. Guest linear memory has no explicit per-instance runtime quota yet; these settings are not a complete resource-admission system. + +## Build, test, and contribute + +Use the checked-in Rust toolchain configuration and lockfile. Native builds need a C/C++ toolchain, `pkg-config`, and OpenSSL development files for the WebAuthn dependency. The standalone storage crate has no Wasmtime dependency; the full runtime includes the AWS SDKs and Wasmtime and is a larger build. + +```bash +# Host artifacts; the runtime includes S3 support without an `aws` feature. +cargo build --locked --release -p enclave-runtime +cargo build --locked --release -p nitro-attestation --features cli --bin nitro-attest + +# Each guest is its own Cargo workspace. +scripts/build-guest.sh http +scripts/build-guest.sh grpc + +# Development-only software passkey client. +cargo build --release -p enclave-runtime --features testing --bin passkey-client +``` + +[`scripts/`](scripts/) contains the same validation entry points CI uses: + +| Command | Coverage / prerequisites | +|---|---| +| `scripts/ci-check.sh` | Formatting, Clippy with warnings denied, workspace library tests, and boot-origin tests. No Docker or built guest required. | +| `scripts/ci-guests.sh` | Builds HTTP/gRPC guests and the test client; tests tasks, held connections, guest dispatch, TLS binding, HTTP/2, gRPC, authentication, and logging. | +| `scripts/ci-storage.sh` | Real MinIO/S3 protocol, retention, remount, corruption, rollback, and competing-claim tests through testcontainers. Requires Docker; runs ignored tests serially. MinIO no longer publishes pullable images, so `scripts/minio-image.sh` builds the pinned release from source on first use (a few minutes), for these tests and the QEMU harness alike. | +| `scripts/ci-e2e.sh` | Guest and storage suites, required local tools, then the QEMU stack. Requires Linux virtualization, Docker, and Nix. | +| `deploy/qemu-nitro/run-e2e.sh` | Direct emulator harness once prerequisites are installed: ACME, authenticated requests, persistence, state origins, and measurement checks. | +| `scripts/ci-bench.sh` | Criterion storage and instance-cost benchmarks; writes `bench.txt`. | +| `scripts/wit-drift.sh` | Checks mirrored guest/runtime WIT definitions for drift. | + +`cargo test --workspace` alone does not stand in for these stages: ignored tests need `--include-ignored`, examples must be built separately, and some binary tests require features. `scripts/ci-sqlite.sh` is an enclave-dependent workload helper and is deliberately not a generic hosted-runner CI gate. + +Validation results belong to a revision; the README does not turn an old test count into a permanent claim that CI is green. + +When extending the runtime, add the capability to the appropriate WIT interface, enforce authority using the invocation's tenant context, and check the matching integration suite. Keep the production image free of the `testing` feature. For stateful changes, test a remount or interrupted operation as well as the successful in-process path. + +## Build and deploy an enclave + +### Reproducible artifacts + +Nix builds the code and configuration inside the enclave. Packer builds the parent AMI; OpenTofu defines deployment resources. The parent is outside the measured image, so these are separate build concerns. + +```bash +# Edit deploy/nix/deployment.nix for the intended deployment first. +nix build .#eif +# result/s3fs.eif and result/pcr.json + +nix build .#guest-release --out-link guest-release +# guest-release/guest.wasm and guest-release/guest-pcr16.json + +# Rebuild and compare the enclave output. +nix build .#eif --rebuild +``` + +The production template selects KMS but does not currently expose dedicated KMS resource fields in `deployment.nix`. Wire `S3FS_KMS_KEY_ID` and `S3FS_MASTER_KEY_PARAMETER` into the measured `runtimeImage.env` in `flake.nix` before building your deployment image; configure `S3FS_ENVIRONMENT` as appropriate. Setting environment variables on the parent service does not inject them into an already-built EIF. These identifiers are configuration, not plaintext key material. + +Image assembly normalizes cpio ordering, ownership, timestamps, and device/inode metadata and controls the EIF builder's timestamp. Dependencies are pinned. The kernel, bootstrap `init`, and NSM module remain pinned AWS prebuilt inputs; a reproducible output is not a claim that every input was independently built from source. See the [Nix build guide](deploy/nix/README.md). + +### Deployment sequence + +1. Set the bucket identities, filesystem ID, region, guest object key, TLS domains, relying party, and enabled capabilities in [`deployment.nix`](deploy/nix/deployment.nix). +2. Build the runtime EIF and guest artifact; retain PCR0 and PCR16 as release outputs. +3. Use [`deploy/tofu`](deploy/tofu/) as the starting point for buckets, retention, parent resources, and logs. Provision the KMS key/policy, SSM parameter access, and related IAM permissions separately; this module does not currently supply the complete KMS/SSM setup. Review measured and infrastructure values together. +4. Upload the approved component to the configured object key and bind KMS release to the approved measurements. +5. Build/configure the parent with [`deploy/ami`](deploy/ami/), then start the enclave and gvproxy services. The parent forwards inbound TLS; it does not terminate application TLS. +6. Verify the boot mode, trusted device sources, actual key-release behavior, TLS/attestation binding, credential path, and persistence from a client pinning the release measurements. + +The roots path needs version listing and reads by version ID as well as retained conditional publication. Object Lock protects versions for their retention period; it is not a general ban on creating delete markers or later versions. + +For public **development** hosts, the QEMU script supports `--domain`, `--port 443`, `--acme-staging`, `--store-bind`, and packing artifacts on one machine for another. A publicly trusted TLS certificate does not make emulator attestations or its development master key equivalent to Nitro. Keep development MinIO credentials off public interfaces and use the full [development guide](docs/DEV_ENCLAVE.md). + +## Use the storage engine directly + +`s3fs-core` can be embedded independently. This complete example creates an in-memory filesystem, commits a file, remounts it, and reads it back. It needs `s3fs-core` and Tokio with macros and a runtime; no NSM, Wasmtime, Docker, or AWS credentials are involved. ```rust use std::sync::Arc; -use s3fs_core::{ - backend::{AwsS3Backend, AwsS3BackendConfig, Backend}, - Config, Fs, OpenFlags, -}; - -#[tokio::main(flavor = "multi_thread")] -async fn main() -> anyhow::Result<()> { - let backend = AwsS3Backend::connect(AwsS3BackendConfig { - bucket: "my-bucket".into(), - region: "us-east-1".into(), - endpoint: None, - access_key_id: Some("AKIA...".into()), - secret_access_key: Some("...".into()), - session_token: None, - force_path_style: false, - request_timeout: std::time::Duration::from_secs(30), - }).await?; - - let fs = Fs::new( - Arc::new(backend) as Arc, - Arc::new(Config::default()), - ); - - let handle = fs.open("data.txt", OpenFlags { read: true, write: true, create: true, ..Default::default() }).await?; - fs.pwrite(&handle, 0, b"hello s3").await?; - fs.sync(&handle).await?; - fs.close(&handle).await?; +use s3fs_core::{backend::memory::MemoryBackend, Config, Fs, MasterSecret, OpenFlags}; + +#[tokio::main(flavor = "current_thread")] +async fn main() -> Result<(), Box> { + let data = Arc::new(MemoryBackend::new()); + let roots = Arc::new(MemoryBackend::new()); + let secret = MasterSecret::from_bytes([7; 32]); // Example only. + let fs_id = [3; 16]; + let config = Arc::new(Config::default()); + + let fs = Fs::create(data.clone(), roots.clone(), &secret, fs_id, config.clone()).await?; + let file = fs.open("/hello.txt", OpenFlags::create_new()).await?; + fs.pwrite(&file, 0, b"hello from encrypted storage").await?; + fs.sync(&file).await?; + fs.close(&file).await?; + drop(fs); + + let fs = Fs::mount(data, roots, &secret, fs_id, config, None).await?; + let file = fs.open("/hello.txt", OpenFlags::read_only()).await?; + let bytes = fs.pread(&file, 0, 64).await?; + assert_eq!(bytes.as_ref(), b"hello from encrypted storage"); + fs.close(&file).await?; Ok(()) } ``` -## Enclave deployment notes +For S3, enable **`s3fs-core`'s** `aws` feature and supply separate `AwsS3Backend` instances for the data and roots buckets. Create the roots bucket with Object Lock support and appropriate retention. Keep the master secret and filesystem ID outside untrusted storage; they select the keys used to verify the store. Use `Fs::create` only for an intentional new filesystem and `Fs::mount` for an existing one. -The original motivation: running this stack inside an [AWS Nitro Enclave](https://aws.amazon.com/ec2/nitro/nitro-enclaves/). Two integration points the enclave consumer is responsible for (these are NOT in this crate): +An embedded consumer owns its provisioning, boot policy, freshness floor, and handle lifecycle. The runtime's origin receipts, tenant gate, and KMS orchestration are higher layers, not automatic side effects of calling the core API. -1. **Vsock-aware HTTP client.** `AwsS3BackendConfig` doesn't currently expose this seam — a future patch will add an `http_client: Option` field that the enclave repo can fill with a vsock-backed `hyper` client. The aws-sdk-s3 default uses TCP. -2. **Cred sourcing via Nitro KMS attestation.** Static creds get passed to `AwsS3BackendConfig` — the parent EC2 ships KMS-encrypted creds over vsock; the enclave decrypts via attested KMS Decrypt and hands the plaintext to the backend. +## Limits and remaining work + +| Area | Current boundary | +|---|---| +| Nitro production validation | Real recipient key release, refusal of substituted measurements, AWS-root attestation, device availability, IMDSv2 networking, and CloudWatch delivery need hardware evidence. | +| Garbage collection | Unimplemented. History, unreachable blocks, and failed-commit slabs accumulate. | +| Single active writer | No distributed writer coordination, scheduler fencing, or transparent multi-enclave failover. | +| Freshness and availability | An external root-sequence floor is needed for the general cold-mount rollback case. The parent/storage operator can deny service. | +| Authorization scope | A passkey approves a route and interaction; it does not approve every payload byte or signing round. | +| Delivery | Inbound streams, background callbacks, held connections, notifications, and logs have different retry/durability contracts. None provides universal exactly-once side effects. | +| Resource isolation | Warm-cache and queue limits exist; per-guest linear-memory quotas and comprehensive public-service admission control do not. | +| SQLite | Rollback journal and serialized access are supported. WAL and cross-instance advisory locking are outside this WASI setup. | +| Upgrade compatibility | The operator/guest must handle application schema and queued-payload evolution. A matching measurement is not a migration. | + +The [roadmap](docs/ROADMAP.md) contains milestone history, including the storage work and planned garbage collection. Some planning prose predates the implemented KMS source; the current implementation and the revision-specific review distinguish code that exists from hardware validation still to be done. + +## Repository guide + +| Path | What to read or change there | +|---|---| +| [`crates/s3fs-core`](crates/s3fs-core/) | Storage backends, cryptography, block store, filesystem, MinIO tests, benchmarks. | +| [`crates/nitro-nsm`](crates/nitro-nsm/) | NSM device access, entropy, PCR operations, attestation requests, self-test binary. | +| [`crates/nitro-attestation`](crates/nitro-attestation/) | Attestation parsing/verification and the independent `nitro-attest` client. | +| [`runtime`](runtime/) | WASI adapter, measured boot, key sources, TLS/auth, tenancy, tasks, streams, notifications, logging, binary and integration tests. | +| [`examples`](examples/) | HTTP, gRPC, and SQLite component workspaces. | +| [`wit`](wit/) | Runtime capability contracts and component worlds. | +| [`deploy/nix`](deploy/nix/) and [`nix`](nix/) | Measured deployment configuration and reproducible EIF assembly. | +| [`deploy/qemu-nitro`](deploy/qemu-nitro/) | Development enclave, emulator self-test, shared startup code, local CA, and full-stack assertions. | +| [`deploy/ami`](deploy/ami/) and [`deploy/tofu`](deploy/tofu/) | Parent AMI/services and AWS infrastructure definitions. | +| [`scripts`](scripts/) | Guest builds, validation stages, MinIO helpers, and benchmarks. | +| [`docs/CLIENT_INTEGRATION.md`](docs/CLIENT_INTEGRATION.md) | Native-client registration, attestation, approval, and certificate pinning. | +| [`docs/STREAMING.md`](docs/STREAMING.md) | Inbound bidirectional streaming and its authorization/lifetime contract. | +| [`docs/BACKGROUND_TASKS.md`](docs/BACKGROUND_TASKS.md) | Durable scheduling, retries, idempotency, and callback authority. | +| [`docs/NOTIFICATIONS.md`](docs/NOTIFICATIONS.md) | Device enrollment and runtime-owned wake delivery. | +| [`docs/COMPATIBILITY.md`](docs/COMPATIBILITY.md) | Filesystem semantics and WASI limitations. | ## License -Apache-2.0. +Workspace packages declare Apache-2.0 in [`Cargo.toml`](Cargo.toml). diff --git a/crates/nitro-attestation/Cargo.toml b/crates/nitro-attestation/Cargo.toml new file mode 100644 index 0000000..c37a5e3 --- /dev/null +++ b/crates/nitro-attestation/Cargo.toml @@ -0,0 +1,50 @@ +[package] +name = "nitro-attestation" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +description = "Parse and verify AWS Nitro Enclaves attestation documents" + +[[bin]] +name = "nitro-attest" +path = "src/bin/nitro-attest.rs" +required-features = ["cli"] + +[features] +default = [] +# The `nitro-attest` command. Off by default so a client embedding the +# verifier does not also compile a TLS client and an argument parser. +cli = ["dep:clap", "dep:rustls", "dep:rustls-pki-types"] +# Builds genuinely signed attestation documents, for tests elsewhere in the +# workspace and for the QEMU harness. Not a stub: it mints a real certificate +# chain and a real ES384 COSE_Sign1, so a verifier cannot pass against it by +# skipping a check. +testing = ["dep:rcgen", "dep:time"] + +[dependencies] +# Deliberately no dependency on `nitro-nsm`. Verification happens on the +# *client*, which has no `/dev/nsm`, is usually not Linux, and must not need a +# device layer to check a signature. +anyhow = { workspace = true } +aws-lc-rs = { workspace = true } +ciborium = "0.2" +coset = "0.4" +x509-parser = "0.18" +base64 = "0.22" +hex = "0.4" + +clap = { version = "4", features = ["derive"], optional = true } +rustls = { version = "0.23", default-features = false, features = ["std", "aws-lc-rs", "tls12"], optional = true } +rustls-pki-types = { version = "1", optional = true } +rcgen = { version = "0.14", default-features = false, features = ["aws_lc_rs", "pem"], optional = true } +# rcgen takes certificate validity as `time::OffsetDateTime` and does not +# re-export the crate, so building a chain needs it directly. +time = { version = "0.3", default-features = false, optional = true } + +[dev-dependencies] +# So plain `cargo test` exercises the document builder without needing +# `--features testing` on the command line. +rcgen = { version = "0.14", default-features = false, features = ["aws_lc_rs", "pem"] } +time = { version = "0.3", default-features = false } diff --git a/crates/nitro-attestation/src/aws-nitro-root-g1.pem b/crates/nitro-attestation/src/aws-nitro-root-g1.pem new file mode 100644 index 0000000..03fd454 --- /dev/null +++ b/crates/nitro-attestation/src/aws-nitro-root-g1.pem @@ -0,0 +1,14 @@ +-----BEGIN CERTIFICATE----- +MIICETCCAZagAwIBAgIRAPkxdWgbkK/hHUbMtOTn+FYwCgYIKoZIzj0EAwMwSTEL +MAkGA1UEBhMCVVMxDzANBgNVBAoMBkFtYXpvbjEMMAoGA1UECwwDQVdTMRswGQYD +VQQDDBJhd3Mubml0cm8tZW5jbGF2ZXMwHhcNMTkxMDI4MTMyODA1WhcNNDkxMDI4 +MTQyODA1WjBJMQswCQYDVQQGEwJVUzEPMA0GA1UECgwGQW1hem9uMQwwCgYDVQQL +DANBV1MxGzAZBgNVBAMMEmF3cy5uaXRyby1lbmNsYXZlczB2MBAGByqGSM49AgEG +BSuBBAAiA2IABPwCVOumCMHzaHDimtqQvkY4MpJzbolL//Zy2YlES1BR5TSksfbb +48C8WBoyt7F2Bw7eEtaaP+ohG2bnUs990d0JX28TcPQXCEPZ3BABIeTPYwEoCWZE +h8l5YoQwTcU/9KNCMEAwDwYDVR0TAQH/BAUwAwEB/zAdBgNVHQ4EFgQUkCW1DdkF +R+eWw5b6cp3PmanfS5YwDgYDVR0PAQH/BAQDAgGGMAoGCCqGSM49BAMDA2kAMGYC +MQCjfy+Rocm9Xue4YnwWmNJVA44fA0P5W2OpYow9OYCVRaEevL8uO1XYru5xtMPW +rfMCMQCi85sWBbJwKKXdS6BptQFuZbT73o/gBh1qUxl/nNr12UO8Yfwr6wPLb+6N +IwLz3/Y= +-----END CERTIFICATE----- \ No newline at end of file diff --git a/crates/nitro-attestation/src/bin/nitro-attest.rs b/crates/nitro-attestation/src/bin/nitro-attest.rs new file mode 100644 index 0000000..2338d9e --- /dev/null +++ b/crates/nitro-attestation/src/bin/nitro-attest.rs @@ -0,0 +1,652 @@ +//! `nitro-attest` — check that an HTTPS endpoint is the enclave you think. +//! +//! ```console +//! $ nitro-attest --url https://enclave.example --pcr0 8cac35ce… \ +//! --guest ./guest.wasm +//! ``` +//! +//! The check that matters is the last one, and it is the reason this tool +//! exists rather than `curl | openssl`: +//! +//! 1. Open a TLS connection and keep the certificate the server presented. +//! 2. Make an ordinary request, quoting a freshly generated nonce in +//! `x-enclave-nonce`. Every response carries a document in +//! `x-enclave-attestation`; there is no separate attestation endpoint. +//! 3. Verify the document's signature and its chain to the AWS Nitro root. +//! 4. Check the document's `user_data` contains **the hash of that same +//! certificate**. +//! +//! Step 4 is what ties the connection to the enclave. Without it, a valid +//! attestation document proves only that *an* enclave exists somewhere; a +//! proxy could fetch a real document from a real enclave and serve it over its +//! own TLS session. With it, the only party who could have produced this +//! document is the one holding the private key for the connection in hand. +//! +//! PKI is deliberately not what establishes trust here. The certificate is +//! accepted whatever a public CA thinks of it, because the attestation is the +//! stronger statement — it says which *code* terminates the connection, which +//! no CA can attest to. A Let's Encrypt certificate still helps browsers, +//! which cannot check attestations; this tool does not need it. +//! +//! ## Which code: two measurements +//! +//! PCR0 measures the runtime image, and the guest is not in it. The runtime +//! fetches the guest at boot, extends PCR16 with its hash and locks the +//! register before it can obtain a key. So `--pcr0` says which runtime, +//! `--guest` or `--pcr16` says which application, and neither says both. +//! +//! ```console +//! $ nitro-attest --measure ./guest.wasm +//! {"sha256": "…", "PCR16": "…"} +//! ``` +//! +//! prints the PCR16 to pin for a component, computed by the same function this +//! tool verifies with — so the value in a key policy and the value a client +//! checks cannot disagree. + +use std::io::{Read, Write}; +use std::net::TcpStream; +use std::path::Path; +use std::sync::Arc; +use std::time::{Duration, SystemTime}; + +use anyhow::{bail, Context, Result}; +use base64::Engine as _; +use clap::Parser; +use nitro_attestation::{ + AttestationHashes, Expectations, Trust, Verified, VerifyOptions, AWS_NITRO_ROOT_G1_PEM, +}; + +/// The request header carrying the client's nonce, base64url without padding. +const NONCE_HEADER: &str = "x-enclave-nonce"; +/// The response header carrying the document, base64. +const ATTESTATION_HEADER: &str = "x-enclave-attestation"; + +#[derive(Parser, Debug)] +#[command(version, about = "Verify an AWS Nitro Enclaves attestation document")] +struct Cli { + /// URL to request, e.g. `https://enclave.example`. The path defaults to + /// `/auth/`. + /// + /// Not every route carries a document: the runtime attests its `/auth/` + /// exchanges, which is where a client identifies the enclave before + /// approving anything, and leaves guest responses alone — a caller has + /// already pinned the certificate by then. `/auth/` needs no credential and + /// reaches no guest; it answers 404 or 405, and is attested either way. + #[arg(long, conflicts_with = "document")] + url: Option, + + /// Read a document from a file instead of fetching one. Base64 or raw + /// COSE_Sign1; the encoding is detected. + #[arg(long, conflicts_with = "url")] + document: Option, + + /// DER of the certificate the connection that produced `--document` was + /// served. Without it a document read from a file proves nothing about any + /// connection, because there is no connection in hand to bind it to. + #[arg(long, requires = "document")] + peer_certificate: Option, + + /// The nonce that request sent, hex. Only for `--document`: a fetched + /// document is checked against a nonce this tool generated itself. + #[arg(long, requires = "document")] + nonce: Option, + + /// Required PCR0, hex. Pins which runtime image is running. Required for + /// verification unless `--unsigned-emulator`. + #[arg(long, required_unless_present_any = ["measure", "unsigned_emulator"])] + pcr0: Option, + + /// Required PCR16, hex. Pins which guest component the runtime measured + /// before it could obtain a key. `--guest` computes it from the component. + /// Required for verification unless `--guest` or `--unsigned-emulator`. + #[arg(long, required_unless_present_any = ["measure", "guest", "unsigned_emulator"])] + pcr16: Option, + + /// Guest component to check against: the PCR16 it measures to, and the + /// document's second hash. + #[arg(long)] + guest: Option, + + /// Print the PCR16 an enclave serving this component attests, and exit. + /// + /// That is the value a KMS key policy pins beside PCR0, and the value a + /// client pins with `--pcr16`. + #[arg(long, value_name = "COMPONENT", conflicts_with_all = ["url", "document"])] + measure: Option, + + /// Trust root, PEM or DER. Defaults to the embedded AWS Nitro root. + #[arg(long)] + trust_root: Option, + + /// Accept a document that does not chain to the AWS root, reporting the + /// result as self-signed. + #[arg(long)] + allow_untrusted_root: bool, + + /// Check the document's *contents* without verifying any signature. + /// + /// Only one producer needs this and it is not a real enclave: QEMU's + /// emulated NSM does not sign its documents at all — its source says + /// "we don't actually sign the data, so we use -1 as the 'alg' value". + /// There is no signature to check and no chain to follow, so nothing here + /// says the document came from an enclave, or from AWS, or from anything + /// other than whoever answered the connection. + /// + /// What it still checks is what the *runtime* put in: the nonce, the PCRs, + /// and whether user_data binds the certificate this connection was + /// served. Those are our code's job and worth testing. The signature is + /// AWS hardware's job and cannot be tested here. + #[arg(long)] + unsigned_emulator: bool, + + /// Reject a document older than this many seconds. + #[arg(long, default_value_t = 300)] + max_age: u64, +} + +fn main() -> std::process::ExitCode { + match run() { + Ok(()) => std::process::ExitCode::SUCCESS, + Err(e) => { + eprintln!("FAIL: {e:#}"); + std::process::ExitCode::FAILURE + } + } +} + +#[cfg(test)] +mod cli_tests { + use super::*; + + #[test] + fn verification_requires_both_measurements_before_reading_or_connecting() { + for source in [ + ["--url", "https://example.test"], + ["--document", "proof.cose"], + ] { + for pins in [ + vec![], + vec!["--pcr0", "ab"], + vec!["--pcr16", "cd"], + vec!["--guest", "guest.wasm"], + ] { + let mut args = vec!["nitro-attest"]; + args.extend(source); + args.extend(pins); + assert_eq!( + Cli::try_parse_from(args).unwrap_err().kind(), + clap::error::ErrorKind::MissingRequiredArgument + ); + } + for guest in [["--pcr16", "cd"], ["--guest", "guest.wasm"]] { + let mut args = vec!["nitro-attest"]; + args.extend(source); + args.extend(["--pcr0", "ab"]); + args.extend(guest); + Cli::try_parse_from(args).unwrap(); + } + } + } + + #[test] + fn measurement_and_explicit_emulator_mode_need_no_pins() { + Cli::try_parse_from(["nitro-attest", "--measure", "guest.wasm"]).unwrap(); + Cli::try_parse_from([ + "nitro-attest", + "--url", + "https://example.test", + "--unsigned-emulator", + ]) + .unwrap(); + } +} + +fn run() -> Result<()> { + let cli = Cli::parse(); + + if let Some(path) = &cli.measure { + let component = + std::fs::read(path).with_context(|| format!("reading {}", path.display()))?; + println!( + "{{\"sha256\": \"{}\", \"PCR16\": \"{}\"}}", + hex::encode(nitro_attestation::sha256(&component)), + hex::encode(nitro_attestation::guest_pcr(&component)) + ); + return Ok(()); + } + + // 20 bytes, the nitriding convention, and enough that an attacker cannot + // have a matching document ready. + let nonce = random_nonce()?; + + let (document, server_certificate) = match (&cli.url, &cli.document) { + (Some(url), _) => { + let fetched = fetch(url, &nonce)?; + (fetched.document, Some(fetched.certificate)) + } + (_, Some(path)) => { + let raw = std::fs::read(path).with_context(|| format!("reading {}", path.display()))?; + let certificate = match &cli.peer_certificate { + Some(path) => Some( + std::fs::read(path).with_context(|| format!("reading {}", path.display()))?, + ), + None => None, + }; + (decode_document(&raw)?, certificate) + } + _ => bail!("pass --url, --document, or --measure"), + }; + + let guest = match &cli.guest { + Some(path) => { + Some(std::fs::read(path).with_context(|| format!("reading {}", path.display()))?) + } + None => None, + }; + let pcr16 = expected_guest_register(cli.pcr16.as_deref(), guest.as_deref())?; + + let trust_root = match &cli.trust_root { + Some(path) => std::fs::read(path).with_context(|| format!("reading {}", path.display()))?, + None => AWS_NITRO_ROOT_G1_PEM.as_bytes().to_vec(), + }; + + let now = SystemTime::now(); + let verified = if cli.unsigned_emulator { + eprintln!( + "WARNING: --unsigned-emulator: no signature and no chain were checked. \ + Nothing here says who produced this document." + ); + Verified { + document: nitro_attestation::parse(&document)?, + trust: Trust::Unsigned, + } + } else { + nitro_attestation::verify( + &document, + &VerifyOptions { + trust_root, + now, + allow_untrusted_root: cli.allow_untrusted_root, + }, + )? + }; + + let mut expectations = Expectations { + max_age: Some(Duration::from_secs(cli.max_age)), + ..Default::default() + }; + match (&cli.url, &cli.nonce) { + // A document we fetched must quote the nonce this process generated. + (Some(_), _) => expectations.nonce = Some(nonce.clone()), + // A document from a file can only be held to a nonce the caller says + // its request sent. Without one there is nothing to compare, and the + // document could be any age the clock allows. + (None, Some(hex)) => { + expectations.nonce = Some(hex::decode(hex.trim()).context("--nonce is not hex")?) + } + (None, None) => {} + } + if let Some(pcr0) = &cli.pcr0 { + expectations = expectations.pcr0(hex::decode(pcr0.trim()).context("--pcr0 is not hex")?); + } + if let Some(pcr16) = pcr16 { + expectations = expectations.pcr(nitro_attestation::PCR_GUEST, pcr16); + } + verified.expect(&expectations, now)?; + + let document = &verified.document; + println!("module {}", document.module_id); + println!( + "PCR0 {}", + document.pcr0_hex().unwrap_or_else(|| "(absent)".into()) + ); + println!( + "PCR16 {}", + document + .pcr(nitro_attestation::PCR_GUEST) + .map(hex::encode) + .unwrap_or_else(|| "(absent: no guest was measured)".into()) + ); + println!("timestamp {:?}", document.timestamp()); + match verified.trust { + // Which root, not just that there was one. `ChainVerified` means the + // presented root equalled the pinned one — and the pinned one is AWS's + // only when `--trust-root` was not given. Saying "the AWS Nitro root" + // unconditionally told an operator verifying against a test root that + // they had verified against AWS, which is the one thing this line is + // for. + Trust::ChainVerified => match &cli.trust_root { + None => println!("chain verified to the AWS Nitro root"), + Some(path) => println!( + "chain verified to the root pinned in {} — NOT AWS's", + path.display() + ), + }, + Trust::SelfSigned => println!("chain SELF-SIGNED — proves nothing about AWS hardware"), + Trust::Unsigned => println!("chain UNSIGNED — nothing verified; contents only"), + } + + check_binding( + document.user_data.as_deref(), + server_certificate.as_deref(), + cli.guest.as_deref().zip(guest.as_deref()), + )?; + + println!("\nOK"); + Ok(()) +} + +/// The PCR16 to require: from `--pcr16`, from `--guest`, or from both when they +/// name the same guest. +fn expected_guest_register( + pinned: Option<&str>, + component: Option<&[u8]>, +) -> Result>> { + let measured = component.map(|c| nitro_attestation::guest_pcr(c).to_vec()); + let Some(pinned) = pinned else { + return Ok(measured); + }; + let pinned = hex::decode(pinned.trim()).context("--pcr16 is not hex")?; + if let Some(measured) = measured { + if measured != pinned { + bail!( + "--pcr16 and --guest name different guests: the component measures to {}", + hex::encode(&measured) + ); + } + } + Ok(Some(pinned)) +} + +/// The step that ties the TLS session to the attested code. +fn check_binding( + user_data: Option<&[u8]>, + server_certificate: Option<&[u8]>, + guest: Option<(&Path, &[u8])>, +) -> Result<()> { + let Some(user_data) = user_data else { + if server_certificate.is_some() { + bail!( + "document carries no user_data, so the TLS certificate is not bound to it \ + and this connection could be terminated by anyone" + ); + } + println!("user_data (absent)"); + return Ok(()); + }; + + let hashes = AttestationHashes::parse(user_data).context( + "document's user_data is not the expected two-hash layout, so the TLS binding \ + cannot be checked", + )?; + println!("tls hash {}", hex::encode(hashes.tls_certificate)); + println!("guest hash {}", hex::encode(hashes.guest)); + + if let Some(der) = server_certificate { + let presented = nitro_attestation::sha256(der); + if presented != hashes.tls_certificate { + bail!( + "the certificate this connection presented ({}) is not the one the enclave \ + attested ({}) — something is terminating TLS in between", + hex::encode(presented), + hex::encode(hashes.tls_certificate) + ); + } + println!("binding the attested certificate is the one serving this connection"); + } + + if let Some((path, bytes)) = guest { + let digest = nitro_attestation::sha256(bytes); + if digest != hashes.guest { + bail!( + "the enclave is serving a different guest: attested {}, local file {}", + hex::encode(hashes.guest), + hex::encode(digest) + ); + } + println!( + "guest matches {} (PCR16 and user_data)", + path.display() + ); + } + + Ok(()) +} + +fn random_nonce() -> Result> { + use aws_lc_rs::rand::SecureRandom; + let mut nonce = vec![0u8; 20]; + aws_lc_rs::rand::SystemRandom::new() + .fill(&mut nonce) + .map_err(|e| anyhow::anyhow!("generating a nonce: {e}"))?; + Ok(nonce) +} + +fn decode_document(raw: &[u8]) -> Result> { + use base64::Engine; + let text = std::str::from_utf8(raw).unwrap_or("").trim(); + if !text.is_empty() + && text.bytes().all(|b| { + b.is_ascii_alphanumeric() + || b == b'+' + || b == b'/' + || b == b'=' + || b.is_ascii_whitespace() + }) + { + let compact: String = text.split_whitespace().collect(); + if let Ok(bytes) = base64::engine::general_purpose::STANDARD.decode(&compact) { + return Ok(bytes); + } + } + Ok(raw.to_vec()) +} + +struct Fetched { + document: Vec, + /// DER of the certificate the server presented. + certificate: Vec, +} + +/// GET the endpoint over TLS, keeping the certificate it presented. +fn fetch(url: &str, nonce: &[u8]) -> Result { + let (host, port, path) = split_url(url)?; + + // Named explicitly: `builder()` resolves the provider from rustls's + // compiled-in features and panics when more than one is present. + let config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .context("selecting TLS protocol versions")? + .dangerous() + .with_custom_certificate_verifier(Arc::new(danger::AcceptAnyServerCert)) + .with_no_client_auth(); + + let server_name = rustls_pki_types::ServerName::try_from(host.clone()) + .with_context(|| format!("{host:?} is not a valid server name"))?; + let mut client = rustls::ClientConnection::new(Arc::new(config), server_name) + .context("starting the TLS session")?; + let mut socket = TcpStream::connect((host.as_str(), port)) + .with_context(|| format!("connecting to {host}:{port}"))?; + let mut tls = rustls::Stream::new(&mut client, &mut socket); + + let request = format!( + "GET {path} HTTP/1.1\r\nHost: {host}\r\n{}: {}\r\nConnection: close\r\n\ + User-Agent: nitro-attest\r\n\r\n", + NONCE_HEADER, + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(nonce) + ); + tls.write_all(request.as_bytes()) + .context("sending the request")?; + + let mut response = Vec::new(); + match tls.read_to_end(&mut response) { + Ok(_) => {} + // Servers commonly drop the connection rather than closing TLS + // cleanly. That is not a failure if the response already arrived. + Err(e) if e.kind() == std::io::ErrorKind::UnexpectedEof && !response.is_empty() => {} + Err(e) => return Err(e).context("reading the response"), + } + + let certificate = client + .peer_certificates() + .and_then(|c| c.first().cloned()) + .context("the server presented no certificate")? + .to_vec(); + + let (status, head, body) = split_response(&response)?; + + // The document rides on the response, whatever the response says. A + // refusal is attested too — a client that only trusted 200s could be + // steered by an unattested 401 into believing the enclave was down. + let encoded = head + .lines() + .find(|l| { + l.to_ascii_lowercase() + .starts_with(&format!("{ATTESTATION_HEADER}:")) + }) + .and_then(|l| l.split_once(':')) + .map(|(_, v)| v.trim().to_string()) + .with_context(|| { + format!( + "no {ATTESTATION_HEADER} header on the HTTP {status} response — \ + attestation is disabled, or something else is answering: {}", + String::from_utf8_lossy(&body).trim() + ) + })?; + + Ok(Fetched { + document: decode_document(encoded.as_bytes())?, + certificate, + }) +} + +fn split_url(url: &str) -> Result<(String, u16, String)> { + let rest = url + .strip_prefix("https://") + .context("--url must start with https://")?; + let (authority, path) = match rest.find('/') { + Some(i) => (&rest[..i], &rest[i..]), + // The probe: attested, needs no credential, reaches no guest. What + // comes back is a refusal, and the document on it is the point. + None => (rest, "/auth/"), + }; + let (host, port) = match authority.rsplit_once(':') { + Some((h, p)) => (h.to_string(), p.parse().context("invalid port")?), + None => (authority.to_string(), 443u16), + }; + Ok((host, port, path.to_string())) +} + +fn split_response(response: &[u8]) -> Result<(u16, String, Vec)> { + let split = response + .windows(4) + .position(|w| w == b"\r\n\r\n") + .context("response has no header terminator")?; + let head = String::from_utf8_lossy(&response[..split]); + let status: u16 = head + .lines() + .next() + .and_then(|l| l.split_whitespace().nth(1)) + .and_then(|s| s.parse().ok()) + .context("response has no status code")?; + + let body = &response[split + 4..]; + let body = if head + .to_ascii_lowercase() + .contains("transfer-encoding: chunked") + { + dechunk(body)? + } else { + body.to_vec() + }; + Ok((status, head.to_string(), body)) +} + +/// Decode `Transfer-Encoding: chunked`. +/// +/// Not optional. A guest response has no `content-length` — the body streams +/// out of the component as it is produced — so hyper chunks it, and an +/// attestation endpoint sitting on the same server may be reached the same +/// way. A client that ignored this would read chunk framing as content. +fn dechunk(body: &[u8]) -> Result> { + let mut out = Vec::new(); + let mut rest = body; + loop { + let end = rest + .windows(2) + .position(|w| w == b"\r\n") + .context("chunked body ended without a chunk header")?; + let header = std::str::from_utf8(&rest[..end]).context("chunk header is not UTF-8")?; + let size = usize::from_str_radix(header.split(';').next().unwrap_or("").trim(), 16) + .context("chunk size is not hex")?; + rest = &rest[end + 2..]; + if size == 0 { + break; + } + if rest.len() < size { + bail!("chunked body is truncated"); + } + out.extend_from_slice(&rest[..size]); + // Skip the chunk's trailing CRLF. + rest = rest.get(size + 2..).unwrap_or(&[]); + } + Ok(out) +} + +mod danger { + use rustls::client::danger::{HandshakeSignatureValid, ServerCertVerified, ServerCertVerifier}; + use rustls::{DigitallySignedStruct, Error, SignatureScheme}; + use rustls_pki_types::{CertificateDer, ServerName, UnixTime}; + + /// Accepts any server certificate. + /// + /// Not an oversight and not a shortcut. Trust here comes from the + /// attestation document, which binds this certificate's hash to a specific + /// enclave image — a strictly stronger statement than any CA makes, and + /// one that holds even for a self-signed certificate. The caller checks + /// that binding; if it does not hold, the connection is rejected there. + /// + /// This is safe *only* because that check is not optional in this tool. + #[derive(Debug)] + pub struct AcceptAnyServerCert; + + impl ServerCertVerifier for AcceptAnyServerCert { + fn verify_server_cert( + &self, + _end_entity: &CertificateDer<'_>, + _intermediates: &[CertificateDer<'_>], + _server_name: &ServerName<'_>, + _ocsp: &[u8], + _now: UnixTime, + ) -> Result { + Ok(ServerCertVerified::assertion()) + } + + fn verify_tls12_signature( + &self, + _message: &[u8], + _cert: &CertificateDer<'_>, + _dss: &DigitallySignedStruct, + ) -> Result { + Ok(HandshakeSignatureValid::assertion()) + } + + fn verify_tls13_signature( + &self, + _message: &[u8], + _cert: &CertificateDer<'_>, + _dss: &DigitallySignedStruct, + ) -> Result { + Ok(HandshakeSignatureValid::assertion()) + } + + fn supported_verify_schemes(&self) -> Vec { + rustls::crypto::aws_lc_rs::default_provider() + .signature_verification_algorithms + .supported_schemes() + } + } +} diff --git a/crates/nitro-attestation/src/lib.rs b/crates/nitro-attestation/src/lib.rs new file mode 100644 index 0000000..bc18b2a --- /dev/null +++ b/crates/nitro-attestation/src/lib.rs @@ -0,0 +1,1010 @@ +//! Parsing and verifying AWS Nitro Enclaves attestation documents. +//! +//! An attestation document is a COSE_Sign1 signed by a key the Nitro +//! hypervisor holds, carrying the enclave's PCR measurements and up to three +//! caller-supplied fields. Verifying one answers a question a TLS handshake +//! cannot: *is the thing I am talking to an enclave running the code I +//! expect?* +//! +//! ```text +//! COSE_Sign1 +//! ├── protected {1: -35} ES384 +//! ├── payload CBOR map +//! │ ├── pcrs locked only 0 the runtime image, 16 its guest +//! │ ├── certificate leaf DER signs this document +//! │ ├── cabundle [DER, …] root first, chains to AWS +//! │ ├── user_data opt bytes ← the TLS certificate hash +//! │ └── nonce opt bytes ← the verifier's freshness challenge +//! └── signature r‖s, 96 bytes +//! ``` +//! +//! This crate deliberately does **not** depend on `nitro-nsm`. Verification +//! happens on the client, which has no `/dev/nsm`, is often not Linux at all, +//! and should not need a device layer to check a signature. +//! +//! ## What [`verify`] checks, and what it does not +//! +//! Checks: the COSE signature against the leaf certificate's key; the leaf +//! chains to the trust root through `cabundle`, each link's signature +//! verified; every certificate's validity window against a supplied time; +//! that intermediates are CAs; and that the root is the one pinned. +//! +//! Does not check: revocation, name constraints, or extended key usage. The +//! chain is linear and its root is pinned by the caller, so path *building* — +//! where most X.509 complexity lives — does not arise. Revocation would need a +//! network fetch from inside a verifier that may have no network. +//! +//! Confirming the document is authentic is only half the job. It says nothing +//! about *what* was attested: a valid document from an enclave running +//! something else is still valid. [`Verified::expect`] is where the caller +//! states what it required — PCR0, the nonce it sent, the certificate it is +//! talking to — and those comparisons are the point. + +#[cfg(any(test, feature = "testing"))] +pub mod testing; + +use std::collections::BTreeMap; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use anyhow::{bail, Context, Result}; +use aws_lc_rs::{digest, signature}; +use x509_parser::prelude::*; + +/// The AWS Nitro Enclaves root, `CN = aws.nitro-enclaves`, valid to 2049. +/// +/// From , +/// whose published SHA-256 is +/// `8cf60e2b2efca96c6a9e71e851d00c1b6991cc09eadbe64a6a1d1b1eb9faff7c`. The +/// certificate's own SHA-256 fingerprint is asserted in the tests below, so a +/// careless edit to this file fails the build rather than silently moving the +/// trust anchor. +pub const AWS_NITRO_ROOT_G1_PEM: &str = include_str!("aws-nitro-root-g1.pem"); + +/// A parsed attestation document payload. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct AttestationDocument { + pub module_id: String, + /// Milliseconds since the Unix epoch, as the *enclave* saw it. + pub timestamp_ms: u64, + pub digest: String, + /// Platform Configuration Registers. PCR0 measures the enclave image. + pub pcrs: BTreeMap>, + /// DER of the certificate whose key signed this document. + pub certificate: Vec, + /// DER certificates from the root down to the leaf's issuer. + pub cabundle: Vec>, + pub public_key: Option>, + pub user_data: Option>, + pub nonce: Option>, +} + +impl AttestationDocument { + pub fn pcr(&self, index: u32) -> Option<&[u8]> { + self.pcrs.get(&index).map(|v| v.as_slice()) + } + + /// PCR0 as lowercase hex — the value a KMS key policy pins. + pub fn pcr0_hex(&self) -> Option { + self.pcr(0).map(hex::encode) + } + + pub fn timestamp(&self) -> SystemTime { + UNIX_EPOCH + Duration::from_millis(self.timestamp_ms) + } +} + +/// How much of the document was checked, so a caller cannot mistake one for +/// the other. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Trust { + /// Signature and certificate chain verified against the pinned root. + ChainVerified, + /// Signature verified against the document's own leaf certificate, but the + /// chain was checked against a caller-supplied root that is not AWS's — so + /// the document is cryptographically self-consistent while proving nothing + /// about AWS hardware. + SelfSigned, + /// Nothing was verified: the contents were read and no signature was + /// checked, because there was none to check. + /// + /// QEMU's emulated NSM produces documents in this state — it does not sign + /// them, and uses an invalid COSE algorithm identifier to say so. A + /// document at this level is worth exactly what the connection it arrived + /// over is worth. + Unsigned, +} + +/// A document that passed [`verify`]. +#[derive(Debug, Clone)] +pub struct Verified { + pub document: AttestationDocument, + pub trust: Trust, +} + +/// What the caller requires the document to say. +/// +/// Every field is optional and every field left `None` is a check *not* +/// performed. A verifier that only calls [`verify`] has established that some +/// enclave signed something; these are what turn that into a statement about +/// this enclave and this connection. +#[derive(Debug, Default, Clone)] +pub struct Expectations { + /// Required PCR values by index. + /// + /// PCR0 pins the runtime image. [`PCR_GUEST`] pins the guest component that + /// runtime measured before it could obtain a key. A client needs both, + /// because the image does not contain the guest and the runtime is what + /// writes the guest register. See [`Expectations::pcr`]. + pub pcrs: BTreeMap>, + /// The nonce the verifier sent. Rejects a replayed document. + pub nonce: Option>, + /// Required `user_data`, byte for byte. + pub user_data: Option>, + /// Maximum age, against the document's own timestamp. + /// + /// Left `None` for a document that is a statement about an *origin* rather + /// than a proof of liveness — a state-origin receipt is supposed to be + /// old, and expiring one would make a filesystem unmountable by the + /// passage of time. Set it for anything a client fetches live. + pub max_age: Option, +} + +impl Expectations { + /// Require `index` to hold `value`. + pub fn pcr(mut self, index: u32, value: impl Into>) -> Self { + self.pcrs.insert(index, value.into()); + self + } + + /// Require the document to have been produced by a given enclave image. + pub fn pcr0(self, value: impl Into>) -> Self { + self.pcr(0, value) + } + + pub fn user_data(mut self, value: impl Into>) -> Self { + self.user_data = Some(value.into()); + self + } + + pub fn nonce(mut self, value: impl Into>) -> Self { + self.nonce = Some(value.into()); + self + } +} + +impl Verified { + /// Apply [`Expectations`], failing on the first that does not hold. + pub fn expect(&self, expectations: &Expectations, now: SystemTime) -> Result<()> { + for (index, want) in &expectations.pcrs { + let got = self + .document + .pcr(*index) + .with_context(|| format!("document carries no PCR{index} to compare"))?; + if got != want.as_slice() { + bail!( + "PCR{index} mismatch: document says {}, expected {}", + hex::encode(got), + hex::encode(want) + ); + } + } + + if let Some(want) = &expectations.nonce { + match &self.document.nonce { + // Without this the document could be one captured earlier from + // the same enclave, which says nothing about now. + None => bail!("document carries no nonce, so it cannot be shown to be fresh"), + Some(got) if got != want => bail!( + "nonce mismatch: document echoes {}, sent {}", + hex::encode(got), + hex::encode(want) + ), + Some(_) => {} + } + } + + if let Some(want) = &expectations.user_data { + match &self.document.user_data { + None => bail!("document carries no user_data"), + Some(got) if got != want => bail!( + "user_data mismatch: document binds {}, expected {}", + hex::encode(got), + hex::encode(want) + ), + Some(_) => {} + } + } + + if let Some(max_age) = expectations.max_age { + let stamped = self.document.timestamp(); + let age = now.duration_since(stamped).unwrap_or_else(|e| e.duration()); + if age > max_age { + bail!("document is {age:?} old, over the {max_age:?} limit"); + } + } + + Ok(()) + } +} + +/// Options for [`verify`]. +pub struct VerifyOptions { + /// PEM or DER of the trust anchor. Defaults to the AWS Nitro root. + pub trust_root: Vec, + /// Time to judge certificate validity windows against. + pub now: SystemTime, + /// Treat a non-AWS root as acceptable, reporting [`Trust::SelfSigned`]. + /// + /// Exists for the QEMU harness, which has no AWS signing key. It must be + /// set deliberately: a verifier that silently accepted whatever root the + /// document arrived with would be checking nothing at all. + pub allow_untrusted_root: bool, +} + +impl Default for VerifyOptions { + fn default() -> Self { + VerifyOptions { + trust_root: AWS_NITRO_ROOT_G1_PEM.as_bytes().to_vec(), + now: SystemTime::now(), + allow_untrusted_root: false, + } + } +} + +/// Parse a COSE_Sign1 attestation document **without verifying anything**. +/// +/// For inspection only. Nothing a document says is worth acting on until +/// [`verify`] has run: the payload is attacker-controlled until its signature +/// is checked. +/// +/// Deliberately more tolerant than [`verify`], because one producer needs it: +/// QEMU's emulated NSM does not sign at all. Its source says so — +/// *"we don't actually sign the data, so we use -1 as the 'alg' value"* — and +/// -1 is not a COSE algorithm identifier, so a strict COSE parser rejects the +/// document before reaching the payload. Reading the payload anyway is what +/// lets the emulator harness check the *contents* the runtime asked for, while +/// [`verify`] keeps refusing it. +pub fn parse(cose: &[u8]) -> Result { + if let Ok(sign1) = parse_cose(cose) { + let payload = sign1 + .payload + .as_ref() + .context("COSE_Sign1 has no payload")?; + return decode_payload(payload); + } + decode_payload(&payload_from_raw_cose(cose)?) +} + +/// Pull the payload out of a COSE_Sign1 without interpreting its headers. +/// +/// The structure is a 4-element array — protected, unprotected, payload, +/// signature — and the payload is element 2 whatever the headers claim. +fn payload_from_raw_cose(cose: &[u8]) -> Result> { + let value: ciborium::Value = ciborium::from_reader(cose).context("document is not CBOR")?; + // Tagged (18) or bare. + let value = match value { + ciborium::Value::Tag(_, inner) => *inner, + other => other, + }; + let array = value.as_array().context("COSE_Sign1 is not a CBOR array")?; + if array.len() != 4 { + bail!("COSE_Sign1 has {} elements, expected 4", array.len()); + } + array[2] + .as_bytes() + .cloned() + .context("COSE_Sign1 payload is not a byte string") +} + +/// Parse and verify a COSE_Sign1 attestation document. +pub fn verify(cose: &[u8], options: &VerifyOptions) -> Result { + let sign1 = parse_cose(cose)?; + let payload = sign1 + .payload + .as_ref() + .context("COSE_Sign1 has no payload")?; + let document = decode_payload(payload)?; + + // Order matters: establish that the leaf is trusted *before* trusting the + // key it carries to check the document's own signature. The other way + // round, a forged document could nominate its own signing certificate. + let trust = verify_chain(&document, options)?; + + let leaf = X509Certificate::from_der(&document.certificate) + .context("parsing the leaf certificate")? + .1; + let key = leaf.public_key().subject_public_key.data.as_ref(); + + // COSE signatures are fixed-width r‖s, not the ASN.1 SEQUENCE that X.509 + // uses. Passing the wrong one here fails on every valid document, which is + // an unhelpfully quiet way to be wrong. + sign1 + .verify_signature(b"", |sig, data| { + signature::UnparsedPublicKey::new(&signature::ECDSA_P384_SHA384_FIXED, key) + .verify(data, sig) + }) + .map_err(|_| anyhow::anyhow!("attestation document signature is not valid"))?; + + Ok(Verified { document, trust }) +} + +fn parse_cose(cose: &[u8]) -> Result { + use coset::{CborSerializable, TaggedCborSerializable}; + // Documents appear both tagged (18) and bare depending on the producer. + coset::CoseSign1::from_tagged_slice(cose) + .or_else(|_| coset::CoseSign1::from_slice(cose)) + .map_err(|e| anyhow::anyhow!("parsing COSE_Sign1: {e:?}")) +} + +fn decode_payload(payload: &[u8]) -> Result { + let value: ciborium::Value = + ciborium::from_reader(payload).context("decoding the attestation payload")?; + let map = value + .as_map() + .context("attestation payload is not a CBOR map")?; + + let get = |name: &str| -> Option<&ciborium::Value> { + map.iter() + .find(|(k, _)| k.as_text() == Some(name)) + .map(|(_, v)| v) + }; + + let text = |v: Option<&ciborium::Value>, name: &str| -> Result { + v.and_then(|v| v.as_text()) + .map(str::to_string) + .with_context(|| format!("attestation payload field {name:?} is missing or not text")) + }; + let bytes = |v: Option<&ciborium::Value>, name: &str| -> Result> { + v.and_then(|v| v.as_bytes()) + .cloned() + .with_context(|| format!("attestation payload field {name:?} is missing or not bytes")) + }; + let optional_bytes = + |v: Option<&ciborium::Value>| -> Option> { v.and_then(|v| v.as_bytes()).cloned() }; + + let module_id = text(get("module_id"), "module_id")?; + let digest = text(get("digest"), "digest")?; + let timestamp_ms = get("timestamp") + .and_then(|v| v.as_integer()) + .and_then(|i| u64::try_from(i).ok()) + .context("attestation payload field \"timestamp\" is missing or not an integer")?; + + let mut pcrs = BTreeMap::new(); + let pcr_map = get("pcrs") + .and_then(|v| v.as_map()) + .context("attestation payload field \"pcrs\" is missing or not a map")?; + for (k, v) in pcr_map { + let index = k + .as_integer() + .and_then(|i| u32::try_from(i).ok()) + .context("PCR index is not an integer")?; + let value = v.as_bytes().cloned().context("PCR value is not bytes")?; + pcrs.insert(index, value); + } + + let certificate = bytes(get("certificate"), "certificate")?; + let cabundle = get("cabundle") + .and_then(|v| v.as_array()) + .context("attestation payload field \"cabundle\" is missing or not an array")? + .iter() + .map(|v| { + v.as_bytes() + .cloned() + .context("cabundle entry is not a byte string") + }) + .collect::>>()?; + + Ok(AttestationDocument { + module_id, + timestamp_ms, + digest, + pcrs, + certificate, + cabundle, + public_key: optional_bytes(get("public_key")), + user_data: optional_bytes(get("user_data")), + nonce: optional_bytes(get("nonce")), + }) +} + +/// Verify `cabundle` chains from the pinned root down to the leaf. +/// +/// `cabundle` is ordered root first. The chain to check is therefore +/// `cabundle` in order, then the leaf. +fn verify_chain(document: &AttestationDocument, options: &VerifyOptions) -> Result { + if document.cabundle.is_empty() { + bail!("attestation document has an empty cabundle, so nothing chains to a root"); + } + + let expected_root = decode_certificate(&options.trust_root)?; + let presented_root = &document.cabundle[0]; + + let trust = if presented_root.as_slice() == expected_root.as_slice() { + Trust::ChainVerified + } else if options.allow_untrusted_root { + Trust::SelfSigned + } else { + bail!( + "attestation document does not chain to the expected root \ + (presented root sha256 {}, expected {})", + sha256_hex(presented_root), + sha256_hex(&expected_root) + ); + }; + + // Root, intermediates, then the leaf: each verified against the one before. + let mut chain: Vec<&[u8]> = document.cabundle.iter().map(|c| c.as_slice()).collect(); + chain.push(&document.certificate); + + for (depth, der) in chain.iter().enumerate() { + let (_, cert) = X509Certificate::from_der(der) + .with_context(|| format!("parsing certificate at depth {depth}"))?; + + let seconds = options + .now + .duration_since(UNIX_EPOCH) + .context("verification time is before the Unix epoch")? + .as_secs(); + let asn1_now = + ASN1Time::from_timestamp(seconds as i64).context("converting the verification time")?; + if !cert.validity().is_valid_at(asn1_now) { + bail!( + "certificate at depth {depth} ({}) is not valid at the given time: {} to {}", + cert.subject(), + cert.validity().not_before, + cert.validity().not_after + ); + } + + // Every certificate except the leaf must be allowed to issue. Without + // this, a leaf could be presented as an intermediate and sign others. + let is_leaf = depth == chain.len() - 1; + if !is_leaf { + let is_ca = cert + .basic_constraints() + .ok() + .flatten() + .map(|bc| bc.value.ca) + .unwrap_or(false); + if !is_ca { + bail!( + "certificate at depth {depth} ({}) is not a CA but has one below it", + cert.subject() + ); + } + } + + if depth == 0 { + // The root is self-signed; its authority comes from being pinned, + // which was settled above. + continue; + } + let (_, issuer) = X509Certificate::from_der(chain[depth - 1]) + .with_context(|| format!("parsing issuer at depth {}", depth - 1))?; + if cert.issuer() != issuer.subject() { + bail!( + "certificate at depth {depth} names issuer {} but follows {}", + cert.issuer(), + issuer.subject() + ); + } + verify_certificate_signature(&cert, &issuer) + .with_context(|| format!("verifying certificate at depth {depth}"))?; + } + + Ok(trust) +} + +/// Check `cert`'s signature against `issuer`'s public key. +/// +/// X.509 ECDSA signatures are DER-encoded `SEQUENCE { r, s }` — the ASN.1 +/// variant — unlike the fixed-width form COSE uses. +fn verify_certificate_signature(cert: &X509Certificate, issuer: &X509Certificate) -> Result<()> { + let algorithm = match cert.signature_algorithm.algorithm.to_id_string().as_str() { + // ecdsa-with-SHA384 + "1.2.840.10045.4.3.3" => &signature::ECDSA_P384_SHA384_ASN1, + // ecdsa-with-SHA256 + "1.2.840.10045.4.3.2" => &signature::ECDSA_P256_SHA256_ASN1, + other => bail!("unsupported certificate signature algorithm {other}"), + }; + + let key = issuer.public_key().subject_public_key.data.as_ref(); + signature::UnparsedPublicKey::new(algorithm, key) + .verify(cert.tbs_certificate.as_ref(), &cert.signature_value.data) + .map_err(|_| anyhow::anyhow!("signature does not verify against the issuer's key")) +} + +/// Accept a trust root as PEM or DER, returning DER. +fn decode_certificate(bytes: &[u8]) -> Result> { + let looks_like_pem = bytes.starts_with(b"-----BEGIN"); + if !looks_like_pem { + return Ok(bytes.to_vec()); + } + let text = std::str::from_utf8(bytes).context("PEM certificate is not UTF-8")?; + let body: String = text + .lines() + .skip_while(|l| !l.starts_with("-----BEGIN")) + .skip(1) + .take_while(|l| !l.starts_with("-----END")) + .collect(); + use base64::Engine; + base64::engine::general_purpose::STANDARD + .decode(body.trim()) + .context("decoding the PEM certificate body") +} + +fn sha256_hex(bytes: &[u8]) -> String { + hex::encode(digest::digest(&digest::SHA256, bytes)) +} + +/// The `user_data` layout nitriding uses, and this runtime follows. +/// +/// Two multihash-prefixed SHA-256 digests, 68 bytes: the TLS certificate the +/// client is talking to, and the guest component being served. Together they +/// answer "is this connection terminated by the code I attested?" — the +/// certificate hash ties the TLS session to the document, and the guest hash +/// says which application was behind it. +/// +/// ```text +/// 0x12 0x20 ‖ sha256(tls leaf DER) ‖ 0x12 0x20 ‖ sha256(guest component) +/// │ └── length, 32 bytes +/// └── multihash code for sha2-256 +/// ``` +/// +/// The prefixes are what make it self-describing: a bare 64-byte blob could +/// not later gain a third hash, or move to SHA-384, without every existing +/// verifier silently misreading it. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct AttestationHashes { + pub tls_certificate: [u8; 32], + pub guest: [u8; 32], +} + +/// Multihash prefix for sha2-256 with a 32-byte digest. +const MULTIHASH_SHA256: [u8; 2] = [0x12, 0x20]; +/// Two prefixed digests. +pub const ATTESTATION_HASHES_LEN: usize = 2 * (2 + 32); + +impl AttestationHashes { + pub fn new(tls_certificate_der: &[u8], guest_component: &[u8]) -> Self { + AttestationHashes { + tls_certificate: sha256(tls_certificate_der), + guest: sha256(guest_component), + } + } + + pub fn serialize(&self) -> Vec { + let mut out = Vec::with_capacity(ATTESTATION_HASHES_LEN); + out.extend_from_slice(&MULTIHASH_SHA256); + out.extend_from_slice(&self.tls_certificate); + out.extend_from_slice(&MULTIHASH_SHA256); + out.extend_from_slice(&self.guest); + out + } + + pub fn parse(bytes: &[u8]) -> Result { + if bytes.len() != ATTESTATION_HASHES_LEN { + bail!( + "user_data is {} bytes, expected {ATTESTATION_HASHES_LEN}", + bytes.len() + ); + } + if bytes[0..2] != MULTIHASH_SHA256 || bytes[34..36] != MULTIHASH_SHA256 { + bail!("user_data does not carry two sha2-256 multihash prefixes"); + } + let mut hashes = AttestationHashes { + tls_certificate: [0u8; 32], + guest: [0u8; 32], + }; + hashes.tls_certificate.copy_from_slice(&bytes[2..34]); + hashes.guest.copy_from_slice(&bytes[36..68]); + Ok(hashes) + } +} + +pub fn sha256(bytes: &[u8]) -> [u8; 32] { + let d = digest::digest(&digest::SHA256, bytes); + let mut out = [0u8; 32]; + out.copy_from_slice(d.as_ref()); + out +} + +/// The register the runtime measures its guest component into. +/// +/// `nitro_nsm::PCR_GUEST` is the device-side copy. It is repeated rather than +/// imported because this crate must not depend on the device layer; a test in +/// `enclave-runtime` holds the two together. +pub const PCR_GUEST: u32 = 16; + +/// What a register holds after exactly one extension with `data`, starting +/// from zero: `SHA384(0⁴⁸ ‖ data)`. +pub fn pcr_after_one_extend(data: &[u8]) -> [u8; 48] { + let mut ctx = digest::Context::new(&digest::SHA384); + ctx.update(&[0u8; 48]); + ctx.update(data); + let mut out = [0u8; 48]; + out.copy_from_slice(ctx.finish().as_ref()); + out +} + +/// The [`PCR_GUEST`] value an enclave serving `component` attests. +/// +/// The runtime extends the register once with `sha256(component)` and locks +/// it, so the value follows from the component alone — which is what lets a +/// client, or a key policy, pin a guest it holds a copy of. +/// +/// It is an identity only **together with PCR0**. The runtime does the +/// extending, so an enclave running someone else's runtime can put this value +/// in the register while running anything at all; PCR0 is what says the +/// runtime that extended it is this one. +pub fn guest_pcr(component: &[u8]) -> [u8; 48] { + pcr_after_one_extend(&sha256(component)) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Against a value computed outside this crate — + /// `{ head -c 48 /dev/zero; printf abc; } | sha384sum` — because this + /// implementation checked against itself would prove nothing. + #[test] + fn one_extension_is_sha384_over_zeros_then_the_data() { + assert_eq!( + hex::encode(pcr_after_one_extend(b"abc")), + "b1c16eb7634112b7c9d5ebd27e62a2d4528bbfcfd68b62d3afd9ecf98e0f413a\ + 84314acce78317fb69fd895155343e09" + ); + } + + #[test] + fn the_guest_register_is_one_extension_with_the_component_hash() { + assert_eq!(guest_pcr(b"guest"), pcr_after_one_extend(&sha256(b"guest"))); + assert_ne!(guest_pcr(b"guest"), guest_pcr(b"another guest")); + } + + /// A document that does not list the guest register fails an expectation on + /// it, rather than passing as though the check were skipped. That is what + /// an enclave that never locked the register looks like. + #[test] + fn a_guest_expectation_fails_on_a_document_without_the_register() { + let chain = crate::testing::TestChain::new().unwrap(); + let verified = verify( + &chain.document(None, None, [0x0e; 48]).unwrap(), + &VerifyOptions { + trust_root: chain.root_der().to_vec(), + now: SystemTime::now(), + allow_untrusted_root: false, + }, + ) + .unwrap(); + let err = verified + .expect( + &Expectations::default().pcr(PCR_GUEST, guest_pcr(b"g").to_vec()), + SystemTime::now(), + ) + .unwrap_err(); + assert!(format!("{err:#}").contains("no PCR16"), "{err:#}"); + } + + /// The published fingerprint of the AWS Nitro Enclaves root. If the + /// embedded PEM is ever edited, this fails rather than quietly moving the + /// trust anchor for every verification. + #[test] + fn the_embedded_root_is_the_published_one() { + let der = decode_certificate(AWS_NITRO_ROOT_G1_PEM.as_bytes()).unwrap(); + assert_eq!( + sha256_hex(&der), + "641a0321a3e244efe456463195d606317ed7cdcc3c1756e09893f3c68f79bb5b" + ); + + let (_, cert) = X509Certificate::from_der(&der).unwrap(); + assert_eq!( + cert.subject().to_string(), + "C=US, O=Amazon, OU=AWS, CN=aws.nitro-enclaves" + ); + assert_eq!(cert.subject(), cert.issuer(), "the root is self-signed"); + } + + #[test] + fn hashes_round_trip() { + let hashes = AttestationHashes::new(b"a certificate", b"a guest"); + let parsed = AttestationHashes::parse(&hashes.serialize()).unwrap(); + assert_eq!(parsed, hashes); + } + + /// The 68-byte layout is a wire format shared with nitriding. Pin it. + #[test] + fn the_hash_layout_is_the_nitriding_one() { + let hashes = AttestationHashes::new(b"cert", b"guest"); + let bytes = hashes.serialize(); + + assert_eq!(bytes.len(), 68); + assert_eq!(&bytes[0..2], &[0x12, 0x20], "sha2-256 multihash prefix"); + assert_eq!(&bytes[2..34], &sha256(b"cert")); + assert_eq!(&bytes[34..36], &[0x12, 0x20]); + assert_eq!(&bytes[36..68], &sha256(b"guest")); + } + + #[test] + fn a_wrong_length_user_data_is_refused() { + assert!(AttestationHashes::parse(&[0u8; 67]).is_err()); + assert!(AttestationHashes::parse(&[0u8; 69]).is_err()); + } + + /// A 68-byte blob with the right length but no prefixes is not ours. + #[test] + fn user_data_without_multihash_prefixes_is_refused() { + let err = AttestationHashes::parse(&[0u8; 68]).unwrap_err(); + assert!(format!("{err:#}").contains("multihash"), "{err:#}"); + } + + #[test] + fn parsing_garbage_is_an_error_not_a_panic() { + assert!(parse(b"").is_err()); + assert!(parse(b"\x84not cose").is_err()); + assert!(parse(&[0u8; 64]).is_err()); + } + + #[test] + fn pem_and_der_roots_decode_alike() { + let der = decode_certificate(AWS_NITRO_ROOT_G1_PEM.as_bytes()).unwrap(); + assert_eq!(decode_certificate(&der).unwrap(), der); + } + + // ---- verification, against genuinely signed documents ------------------- + // + // These use `testing::TestChain`, which mints a real P-384 chain and signs + // a real ES384 COSE_Sign1. A verifier that skipped the signature, the + // chain, or the root pin would pass the happy-path test and fail the rest, + // which is the reason the rest exist. + + use crate::testing::TestChain; + + fn options_for(chain: &TestChain) -> VerifyOptions { + VerifyOptions { + trust_root: chain.root_der().to_vec(), + now: SystemTime::now(), + allow_untrusted_root: false, + } + } + + #[test] + fn a_well_formed_document_verifies() { + let chain = TestChain::new().unwrap(); + let cose = chain + .document(Some(b"bound".to_vec()), Some(b"fresh".to_vec()), [0xab; 48]) + .unwrap(); + + let verified = verify(&cose, &options_for(&chain)).unwrap(); + assert_eq!(verified.trust, Trust::ChainVerified); + assert_eq!(verified.document.user_data.as_deref(), Some(&b"bound"[..])); + assert_eq!(verified.document.nonce.as_deref(), Some(&b"fresh"[..])); + assert_eq!(verified.document.pcr(0).unwrap(), [0xab; 48]); + assert_eq!(verified.document.pcr0_hex().unwrap(), "ab".repeat(48)); + } + + /// The single most important negative test: swapping the payload for + /// another must not verify. If it does, every other check is decoration. + #[test] + fn a_tampered_payload_does_not_verify() { + let chain = TestChain::new().unwrap(); + let honest = chain + .document(Some(b"honest".to_vec()), None, [0x01; 48]) + .unwrap(); + let forged_payload = testing::encode_payload(&AttestationDocument { + user_data: Some(b"forged".to_vec()), + ..parse(&honest).unwrap() + }); + + // Same signature, different payload. + let mut sign1 = parse_cose(&honest).unwrap(); + sign1.payload = Some(forged_payload); + use coset::CborSerializable; + let tampered = sign1.to_vec().unwrap(); + + let err = verify(&tampered, &options_for(&chain)).unwrap_err(); + assert!(format!("{err:#}").contains("signature"), "{err:#}"); + } + + /// A document signed by a chain rooted somewhere else must not verify + /// against our root, however well formed it is. This is the check that + /// stops anyone with a certificate generator from minting attestations. + #[test] + fn a_document_from_another_root_is_refused() { + let ours = TestChain::new().unwrap(); + let theirs = TestChain::new().unwrap(); + let cose = theirs.document(None, None, [0x02; 48]).unwrap(); + + let err = verify(&cose, &options_for(&ours)).unwrap_err(); + assert!(format!("{err:#}").contains("root"), "{err:#}"); + } + + /// The escape hatch the QEMU harness needs, and the reason it reports a + /// different `Trust`: the document is self-consistent but proves nothing + /// about AWS. + #[test] + fn an_untrusted_root_is_accepted_only_when_asked_and_is_labelled() { + let theirs = TestChain::new().unwrap(); + let cose = theirs.document(None, None, [0x03; 48]).unwrap(); + + let options = VerifyOptions { + trust_root: AWS_NITRO_ROOT_G1_PEM.as_bytes().to_vec(), + now: SystemTime::now(), + allow_untrusted_root: true, + }; + let verified = verify(&cose, &options).unwrap(); + assert_eq!( + verified.trust, + Trust::SelfSigned, + "a non-AWS root must never be reported as chain-verified" + ); + } + + #[test] + fn an_expired_chain_is_refused() { + let long_ago = SystemTime::now() - Duration::from_secs(3600 * 24 * 30); + let chain = TestChain::with_validity( + long_ago, + long_ago + Duration::from_secs(3600), // expired 30 days ago + ) + .unwrap(); + let cose = chain.document(None, None, [0x04; 48]).unwrap(); + + let err = verify(&cose, &options_for(&chain)).unwrap_err(); + assert!(format!("{err:#}").contains("not valid at"), "{err:#}"); + } + + /// The leaf's own certificate replaced by one from a different chain: the + /// signature would verify against the substituted key, so only the chain + /// check catches it. + #[test] + fn a_leaf_from_another_chain_is_refused() { + let ours = TestChain::new().unwrap(); + let theirs = TestChain::new().unwrap(); + + let cose = ours + .document_from(AttestationDocument { + certificate: theirs.leaf.clone(), + ..parse(&ours.document(None, None, [0x05; 48]).unwrap()).unwrap() + }) + .unwrap(); + + assert!(verify(&cose, &options_for(&ours)).is_err()); + } + + #[test] + fn an_empty_cabundle_is_refused() { + let chain = TestChain::new().unwrap(); + let cose = chain + .document_from(AttestationDocument { + cabundle: vec![], + ..parse(&chain.document(None, None, [0x06; 48]).unwrap()).unwrap() + }) + .unwrap(); + + let err = verify(&cose, &options_for(&chain)).unwrap_err(); + assert!(format!("{err:#}").contains("cabundle"), "{err:#}"); + } + + // ---- expectations ------------------------------------------------------- + + fn verified_with( + user_data: Option>, + nonce: Option>, + pcr0: [u8; 48], + ) -> Verified { + let chain = TestChain::new().unwrap(); + let cose = chain.document(user_data, nonce, pcr0).unwrap(); + verify(&cose, &options_for(&chain)).unwrap() + } + + #[test] + fn expectations_hold_when_the_document_matches() { + let verified = verified_with(Some(b"ud".to_vec()), Some(b"n".to_vec()), [0x07; 48]); + verified + .expect( + &Expectations { + pcrs: [(0u32, vec![0x07; 48])].into(), + nonce: Some(b"n".to_vec()), + user_data: Some(b"ud".to_vec()), + max_age: Some(Duration::from_secs(60)), + }, + SystemTime::now(), + ) + .unwrap(); + } + + #[test] + fn a_different_pcr0_is_rejected() { + let verified = verified_with(None, None, [0x08; 48]); + let err = verified + .expect( + &Expectations { + pcrs: [(0u32, vec![0x09; 48])].into(), + ..Default::default() + }, + SystemTime::now(), + ) + .unwrap_err(); + assert!(format!("{err:#}").contains("PCR0 mismatch"), "{err:#}"); + } + + /// A document with no nonce cannot be shown to be fresh, and must not pass + /// as though the check were simply skipped. + #[test] + fn a_missing_nonce_fails_a_nonce_expectation() { + let verified = verified_with(None, None, [0x0a; 48]); + let err = verified + .expect( + &Expectations { + nonce: Some(b"sent".to_vec()), + ..Default::default() + }, + SystemTime::now(), + ) + .unwrap_err(); + assert!(format!("{err:#}").contains("no nonce"), "{err:#}"); + } + + #[test] + fn a_replayed_nonce_is_rejected() { + let verified = verified_with(None, Some(b"old".to_vec()), [0x0b; 48]); + assert!(verified + .expect( + &Expectations { + nonce: Some(b"new".to_vec()), + ..Default::default() + }, + SystemTime::now(), + ) + .is_err()); + } + + #[test] + fn a_stale_document_is_rejected() { + let verified = verified_with(None, None, [0x0c; 48]); + let err = verified + .expect( + &Expectations { + max_age: Some(Duration::from_secs(1)), + ..Default::default() + }, + SystemTime::now() + Duration::from_secs(600), + ) + .unwrap_err(); + assert!(format!("{err:#}").contains("old"), "{err:#}"); + } + + /// The end-to-end shape this milestone exists for: a client holding a TLS + /// certificate checks that the enclave bound *that* certificate. + #[test] + fn a_certificate_binding_round_trips_through_user_data() { + let tls_cert = b"the DER a client saw in the handshake"; + let guest = b"the guest component"; + let hashes = AttestationHashes::new(tls_cert, guest); + + let verified = verified_with(Some(hashes.serialize()), None, [0x0d; 48]); + verified + .expect( + &Expectations { + user_data: Some(hashes.serialize()), + ..Default::default() + }, + SystemTime::now(), + ) + .unwrap(); + + // And a client seeing a *different* certificate must not be satisfied. + let other = AttestationHashes::new(b"a certificate from a proxy", guest); + assert!(verified + .expect( + &Expectations { + user_data: Some(other.serialize()), + ..Default::default() + }, + SystemTime::now(), + ) + .is_err()); + } +} diff --git a/crates/nitro-attestation/src/testing.rs b/crates/nitro-attestation/src/testing.rs new file mode 100644 index 0000000..1860b44 --- /dev/null +++ b/crates/nitro-attestation/src/testing.rs @@ -0,0 +1,255 @@ +//! Building genuinely signed attestation documents. +//! +//! Not a stub. This mints a real P-384 certificate chain and a real ES384 +//! COSE_Sign1 over a real CBOR payload, so a verifier cannot pass against it +//! by skipping a step — the way it would against a hand-written blob. That +//! matters because [`crate::verify`] is the one function here whose failure +//! mode is silent acceptance. +//! +//! What it cannot do is sign with AWS's key. Documents from here chain to a +//! root generated on the spot, so verifying one requires +//! [`crate::VerifyOptions::allow_untrusted_root`] and reports +//! [`crate::Trust::SelfSigned`]. That is the same footing the QEMU harness is +//! on, and the reason [`crate::Trust`] distinguishes the two cases at all. + +use std::collections::BTreeMap; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use anyhow::{Context, Result}; +use aws_lc_rs::rand::SystemRandom; +use aws_lc_rs::signature::{self, EcdsaKeyPair, KeyPair}; +use coset::{CborSerializable, CoseSign1Builder, HeaderBuilder}; +use rcgen::{ + BasicConstraints, CertificateParams, DistinguishedName, DnType, IsCa, Issuer, + KeyPair as RcgenKeyPair, KeyUsagePurpose, +}; + +use crate::AttestationDocument; + +/// A certificate chain plus the leaf's signing key. +#[derive(Debug)] +pub struct TestChain { + /// DER, root first — the `cabundle` layout. + pub cabundle: Vec>, + /// DER of the leaf that signs documents. + pub leaf: Vec, + leaf_key_pkcs8: Vec, +} + +impl TestChain { + /// A root → intermediate → leaf chain, mirroring what a real NSM presents. + pub fn new() -> Result { + Self::with_validity( + SystemTime::now() - Duration::from_secs(3600), + SystemTime::now() + Duration::from_secs(3600 * 24), + ) + } + + pub fn with_validity(not_before: SystemTime, not_after: SystemTime) -> Result { + let root_key = RcgenKeyPair::generate_for(&rcgen::PKCS_ECDSA_P384_SHA384)?; + let mut root_params = ca_params("test-nitro-root", not_before, not_after); + root_params.is_ca = IsCa::Ca(BasicConstraints::Unconstrained); + let root = root_params.self_signed(&root_key)?; + let root_issuer = Issuer::new(root_params, root_key); + + let inter_key = RcgenKeyPair::generate_for(&rcgen::PKCS_ECDSA_P384_SHA384)?; + let mut inter_params = ca_params("test-nitro-intermediate", not_before, not_after); + inter_params.is_ca = IsCa::Ca(BasicConstraints::Constrained(0)); + let intermediate = inter_params.signed_by(&inter_key, &root_issuer)?; + let inter_issuer = Issuer::new(inter_params, inter_key); + + let leaf_key = RcgenKeyPair::generate_for(&rcgen::PKCS_ECDSA_P384_SHA384)?; + let mut leaf_params = ca_params("test-enclave-leaf", not_before, not_after); + leaf_params.is_ca = IsCa::NoCa; + let leaf = leaf_params.signed_by(&leaf_key, &inter_issuer)?; + + Ok(TestChain { + cabundle: vec![root.der().to_vec(), intermediate.der().to_vec()], + leaf: leaf.der().to_vec(), + leaf_key_pkcs8: leaf_key.serialize_der(), + }) + } + + /// PEM of the root, for [`crate::VerifyOptions::trust_root`]. + pub fn root_der(&self) -> &[u8] { + &self.cabundle[0] + } + + /// Build a document with these fields and sign it with the leaf key. + pub fn document( + &self, + user_data: Option>, + nonce: Option>, + pcr0: [u8; 48], + ) -> Result> { + let mut pcrs = BTreeMap::new(); + pcrs.insert(0u32, pcr0.to_vec()); + pcrs.insert(1u32, vec![0x11; 48]); + pcrs.insert(2u32, vec![0x22; 48]); + self.document_with_pcrs(user_data, nonce, pcrs) + } + + /// The same, listing exactly the registers given. + /// + /// A real document lists the registers that are *locked*, so this is how a + /// test says which those are — with the guest register, or deliberately + /// without it. + pub fn document_with_pcrs( + &self, + user_data: Option>, + nonce: Option>, + pcrs: BTreeMap>, + ) -> Result> { + self.document_from(AttestationDocument { + module_id: "i-0test-enc0000000000".to_string(), + timestamp_ms: SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_millis() as u64, + digest: "SHA384".to_string(), + pcrs, + certificate: self.leaf.clone(), + cabundle: self.cabundle.clone(), + public_key: None, + user_data, + nonce, + }) + } + + /// Sign an arbitrary document, so tests can build malformed ones. + pub fn document_from(&self, document: AttestationDocument) -> Result> { + let payload = encode_payload(&document); + + let key = EcdsaKeyPair::from_pkcs8( + &signature::ECDSA_P384_SHA384_FIXED_SIGNING, + &self.leaf_key_pkcs8, + ) + .map_err(|e| anyhow::anyhow!("loading the leaf key: {e}"))?; + let rng = SystemRandom::new(); + + // ES384. The `_FIXED_` algorithm produces r‖s, which is what COSE + // wants — the ASN.1 variant would verify nowhere. + let protected = HeaderBuilder::new() + .algorithm(coset::iana::Algorithm::ES384) + .build(); + let sign1 = CoseSign1Builder::new() + .protected(protected) + .payload(payload) + .try_create_signature(b"", |data| { + key.sign(&rng, data) + .map(|s| s.as_ref().to_vec()) + .map_err(|e| anyhow::anyhow!("signing: {e}")) + })? + .build(); + + sign1 + .to_vec() + .map_err(|e| anyhow::anyhow!("serialising COSE_Sign1: {e:?}")) + } + + /// The leaf's public key, for tests that need to check it directly. + pub fn leaf_public_key(&self) -> Result> { + let key = EcdsaKeyPair::from_pkcs8( + &signature::ECDSA_P384_SHA384_FIXED_SIGNING, + &self.leaf_key_pkcs8, + ) + .map_err(|e| anyhow::anyhow!("loading the leaf key: {e}"))?; + Ok(key.public_key().as_ref().to_vec()) + } +} + +fn ca_params(name: &str, not_before: SystemTime, not_after: SystemTime) -> CertificateParams { + let mut params = CertificateParams::default(); + let mut dn = DistinguishedName::new(); + dn.push(DnType::CommonName, name); + params.distinguished_name = dn; + params.not_before = to_offset(not_before); + params.not_after = to_offset(not_after); + params.key_usages = vec![ + KeyUsagePurpose::DigitalSignature, + KeyUsagePurpose::KeyCertSign, + KeyUsagePurpose::CrlSign, + ]; + params +} + +fn to_offset(t: SystemTime) -> time::OffsetDateTime { + let secs = t.duration_since(UNIX_EPOCH).unwrap().as_secs() as i64; + time::OffsetDateTime::from_unix_timestamp(secs).expect("representable timestamp") +} + +/// Encode a document the way the NSM does, so the decoder is tested against +/// the real wire shape rather than against itself. +pub fn encode_payload(document: &AttestationDocument) -> Vec { + let mut fields = vec![ + ( + ciborium::Value::Text("module_id".into()), + ciborium::Value::Text(document.module_id.clone()), + ), + ( + ciborium::Value::Text("digest".into()), + ciborium::Value::Text(document.digest.clone()), + ), + ( + ciborium::Value::Text("timestamp".into()), + ciborium::Value::Integer(document.timestamp_ms.into()), + ), + ( + ciborium::Value::Text("pcrs".into()), + ciborium::Value::Map( + document + .pcrs + .iter() + .map(|(k, v)| { + ( + ciborium::Value::Integer((*k).into()), + ciborium::Value::Bytes(v.clone()), + ) + }) + .collect(), + ), + ), + ( + ciborium::Value::Text("certificate".into()), + ciborium::Value::Bytes(document.certificate.clone()), + ), + ( + ciborium::Value::Text("cabundle".into()), + ciborium::Value::Array( + document + .cabundle + .iter() + .map(|c| ciborium::Value::Bytes(c.clone())) + .collect(), + ), + ), + ]; + + // The real device emits these as null when absent rather than omitting + // them; the decoder must tolerate either. + for (name, value) in [ + ("public_key", &document.public_key), + ("user_data", &document.user_data), + ("nonce", &document.nonce), + ] { + fields.push(( + ciborium::Value::Text(name.into()), + match value { + Some(bytes) => ciborium::Value::Bytes(bytes.clone()), + None => ciborium::Value::Null, + }, + )); + } + + let mut out = Vec::new(); + ciborium::into_writer(&ciborium::Value::Map(fields), &mut out) + .expect("writing to a Vec cannot fail"); + out +} + +/// Read a document's payload without verifying it, for tests that tamper. +pub fn payload_of(cose: &[u8]) -> Result> { + let sign1 = coset::CoseSign1::from_slice(cose) + .map_err(|e| anyhow::anyhow!("parsing COSE_Sign1: {e:?}"))?; + sign1.payload.context("COSE_Sign1 has no payload") +} diff --git a/crates/nitro-nsm/Cargo.toml b/crates/nitro-nsm/Cargo.toml new file mode 100644 index 0000000..d2c198e --- /dev/null +++ b/crates/nitro-nsm/Cargo.toml @@ -0,0 +1,28 @@ +[package] +name = "nitro-nsm" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +description = "The AWS Nitro Security Module through /dev/nsm: raw CBOR ioctl, GetRandom" + +[features] +default = [] +# Exposes the in-memory fake so dependents can test against it. +testing = [] + +[[bin]] +name = "nsm-selftest" +path = "src/bin/nsm-selftest.rs" + +# Pure Rust on purpose. This crate goes into an enclave ramdisk as a +# statically linked musl binary, so a C dependency anywhere here would mean +# a cross-compiling toolchain for something that needs two syscalls. +[dependencies] +# SHA-384, for computing what a PCR extension would produce. Same crypto stack +# as everything else rather than a second hash implementation. +aws-lc-rs = { workspace = true } +anyhow = { workspace = true } +ciborium = "0.2" +rustix = { version = "1", features = ["fs"] } diff --git a/crates/nitro-nsm/src/bin/nsm-selftest.rs b/crates/nitro-nsm/src/bin/nsm-selftest.rs new file mode 100644 index 0000000..18956b9 --- /dev/null +++ b/crates/nitro-nsm/src/bin/nsm-selftest.rs @@ -0,0 +1,81 @@ +//! Prove the Nitro Security Module answers, from inside an enclave. +//! +//! Deliberately tiny and pure Rust: this is what goes into the enclave ramdisk +//! as a statically linked musl binary, where the full runtime's C dependencies +//! and dynamic linking would be a problem for no benefit. It touches no +//! storage and needs no network — the NSM is the only thing under test. +//! +//! Prints `NSM-SELFTEST-OK` on success and `NSM-SELFTEST-FAIL: …` otherwise, +//! sentinels the harness greps for in the console log. + +use std::time::Instant; + +use nitro_nsm::{Nsm, NsmDevice, DEFAULT_NSM_DEVICE}; + +fn main() { + let device = std::env::args() + .nth(1) + .unwrap_or_else(|| DEFAULT_NSM_DEVICE.to_string()); + match run(&device) { + Ok(()) => println!("NSM-SELFTEST-OK"), + Err(e) => { + println!("NSM-SELFTEST-FAIL: {e:#}"); + std::process::exit(1); + } + } +} + +fn run(device: &str) -> anyhow::Result<()> { + println!("opening {device}"); + let nsm = NsmDevice::open(device)?; + println!("{}", nsm.describe()); + + let mut first = [0u8; 64]; + let started = Instant::now(); + nsm.get_random(&mut first)?; + let one_read = started.elapsed(); + + let mut second = [0u8; 64]; + nsm.get_random(&mut second)?; + + // The failures an emulated or half-wired device actually produces. + if first.iter().all(|&b| b == 0) { + anyhow::bail!("GetRandom returned all zeros"); + } + if first.iter().all(|&b| b == first[0]) { + anyhow::bail!("GetRandom returned the constant byte {:#04x}", first[0]); + } + if first == second { + anyhow::bail!("two draws were identical; the device is not advancing"); + } + + // Larger than one 256-byte device answer, so the chunk loop is exercised. + let mut bulk = vec![0u8; 4096]; + let bulk_started = Instant::now(); + nsm.get_random(&mut bulk)?; + let bulk_read = bulk_started.elapsed(); + + let mut histogram = [0u32; 256]; + for &b in &bulk { + histogram[b as usize] += 1; + } + let peak = histogram.iter().copied().max().unwrap_or(0); + if peak > 80 { + anyhow::bail!("byte histogram peaks at {peak} of 4096; the device looks stuck"); + } + + println!( + "sample {}", + first[..16] + .iter() + .map(|b| format!("{b:02x}")) + .collect::() + ); + println!("histogram peak {peak} of 4096 (mean 16)"); + println!( + "read cost {:.1} us for 64 bytes, {:.1} us for 4096", + one_read.as_secs_f64() * 1e6, + bulk_read.as_secs_f64() * 1e6 + ); + Ok(()) +} diff --git a/crates/nitro-nsm/src/lib.rs b/crates/nitro-nsm/src/lib.rs new file mode 100644 index 0000000..256e592 --- /dev/null +++ b/crates/nitro-nsm/src/lib.rs @@ -0,0 +1,913 @@ +//! The Nitro Security Module, through `/dev/nsm`. +//! +//! The NSM is the enclave's root of trust. It produces attestation documents, +//! holds the PCRs, and — the part used here — generates random bytes. Its +//! userspace interface is one ioctl carrying a CBOR request and returning a +//! CBOR response, defined by `uapi/linux/nsm.h`: +//! +//! ```c +//! struct nsm_iovec { __u64 addr; __u64 len; }; +//! struct nsm_raw { struct nsm_iovec request; struct nsm_iovec response; }; +//! #define NSM_IOCTL_RAW _IOWR(0x0A, 0x0, struct nsm_raw) +//! ``` +//! +//! Only [`NsmDevice::request`] knows about the ioctl; everything above it deals +//! in CBOR payloads. That is the seam attestation will reuse — its requests are +//! different bytes through the same call. +//! +//! ## Two things worth knowing before debugging this +//! +//! The raw ioctl requires **`CAP_SYS_ADMIN`**, per the comment in the kernel +//! header. An unprivileged process gets `EPERM` from a device that exists and +//! is readable, which is a confusing failure if you are not expecting it. +//! +//! `response.len` is **in/out**: the caller sets the buffer capacity and the +//! driver overwrites it with how much it actually wrote. Passing a fresh +//! `nsm_raw` per call rather than reusing one is deliberate. + +use std::fmt; +use std::fs::File; +use std::os::fd::{AsFd, OwnedFd}; +use std::path::{Path, PathBuf}; + +use anyhow::{bail, Context, Result}; +use rustix::ioctl; + +/// The conventional device node for the NSM. +pub const DEFAULT_NSM_DEVICE: &str = "/dev/nsm"; + +/// `NSM_REQUEST_MAX_SIZE`. +const REQUEST_MAX: usize = 0x1000; +/// `NSM_RESPONSE_MAX_SIZE`. +const RESPONSE_MAX: usize = 0x3000; + +/// `struct nsm_iovec`. +#[repr(C)] +#[derive(Debug, Clone, Copy)] +struct NsmIovec { + addr: u64, + len: u64, +} + +/// `struct nsm_raw`. +#[repr(C)] +#[derive(Debug, Clone, Copy)] +struct NsmRaw { + request: NsmIovec, + response: NsmIovec, +} + +/// `_IOWR(NSM_MAGIC, 0x0, struct nsm_raw)`. +const NSM_IOCTL_RAW: ioctl::Opcode = ioctl::opcode::read_write::(0x0A, 0x0); + +/// What goes into an attestation document, beyond what the NSM puts there +/// itself (the PCRs, the module id, the timestamp, the signing certificate). +/// +/// All three fields are optional and all three are attacker-visible; none is a +/// secret. Their value is that the NSM copies them verbatim into a document it +/// signs, so a verifier who trusts the signature can trust that *this enclave* +/// asserted them. +#[derive(Debug, Default, Clone, PartialEq, Eq)] +pub struct AttestationRequest { + /// Application-chosen bytes. This is where a TLS certificate hash goes, so + /// a client can tie the connection it holds to the code it attested. + pub user_data: Option>, + /// Caller-chosen bytes, echoed back. A verifier supplies a fresh one so a + /// replayed document from an earlier boot cannot be passed off as current. + pub nonce: Option>, + /// A public key the enclave generated, for a verifier to encrypt to. Used + /// by KMS `Decrypt` with `Recipient`; unused here. + pub public_key: Option>, +} + +impl AttestationRequest { + /// An attestation binding `user_data`, with a caller-supplied `nonce`. + pub fn with_user_data(user_data: Vec) -> Self { + AttestationRequest { + user_data: Some(user_data), + ..Default::default() + } + } + + pub fn nonce(mut self, nonce: Vec) -> Self { + self.nonce = Some(nonce); + self + } +} + +/// What an NSM can do, so tests can substitute a fake. +pub trait Nsm: Send + Sync + fmt::Debug { + /// Fill `buf` with random bytes. + fn get_random(&self, buf: &mut [u8]) -> Result<()>; + + /// Ask for an attestation document. + /// + /// Returns the raw COSE_Sign1 bytes, unparsed and unverified. Parsing + /// belongs to `nitro-attestation`, which a *client* needs without this + /// crate's Linux device layer — the whole point of a document is that it + /// is checked somewhere else. + fn attest(&self, request: &AttestationRequest) -> Result>; + + /// Read a Platform Configuration Register. + /// + /// Returns the value and whether it is locked. PCRs 0–15 are locked by the + /// hypervisor at boot — 0–2 hold the image measurements — and 16–31 are the + /// enclave's to extend and then lock. + fn describe_pcr(&self, index: u16) -> Result; + + /// Extend a PCR, returning its new value. + /// + /// `new = SHA384(old ‖ data)`, and it cannot be undone for the life of the + /// enclave — which is the property that makes it useful. An enclave that + /// extends a register has permanently committed to whatever it extended + /// with. + /// + /// Fails with a read-only error on a locked register. + fn extend_pcr(&self, index: u16, data: &[u8]) -> Result>; + + /// Lock a PCR, so nothing extends it again for the life of the enclave. + /// + /// Not a formality. An attestation document carries **only locked** + /// registers, so a register that was extended and never locked appears in + /// no document, and no key policy or client can see it. Locking is also + /// what makes the value final: until then, code running later could extend + /// it again. + /// + /// Fails with a read-only error on a register that is already locked. + fn lock_pcr(&self, index: u16) -> Result<()>; + + /// Short description for logs. + fn describe(&self) -> String; +} + +/// A Platform Configuration Register. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Pcr { + pub locked: bool, + pub value: Vec, +} + +/// The register the runtime measures its guest component into. +/// +/// The first one the enclave may use: 0–15 are locked by the hypervisor at +/// boot. The runtime extends it exactly once, with `sha256(component)`, and +/// locks it before asking for any key, so it is reserved — anything else +/// extending it makes the runtime refuse to start rather than attest a value +/// that means something different. +/// +/// `nitro_attestation::PCR_GUEST` is the verifier's copy of this number. +pub const PCR_GUEST: u16 = 16; + +/// What a PCR holds before anything extends it: 48 zero bytes. +pub const PCR_ZERO: [u8; 48] = [0u8; 48]; + +/// The value a register would hold after extending `current` with `data`. +/// +/// Lets the runtime check what the device reported without trusting anything +/// but the arithmetic. +pub fn pcr_extend(current: &[u8], data: &[u8]) -> Vec { + use aws_lc_rs::digest; + let mut ctx = digest::Context::new(&digest::SHA384); + ctx.update(current); + ctx.update(data); + ctx.finish().as_ref().to_vec() +} + +/// The real device. +pub struct NsmDevice { + device: PathBuf, + fd: OwnedFd, +} + +impl NsmDevice { + /// Open the device and prove it answers, so a device that exists but is + /// unusable fails here rather than on the guest's first request for + /// randomness. + pub fn open(device: impl AsRef) -> Result { + let device = device.as_ref().to_path_buf(); + let file = File::open(&device) + .with_context(|| format!("opening NSM device {}", device.display()))?; + let nsm = NsmDevice { + device, + fd: OwnedFd::from(file), + }; + + let mut probe = [0u8; 32]; + nsm.get_random(&mut probe).with_context(|| { + format!( + "NSM device {} did not answer GetRandom (the raw ioctl needs CAP_SYS_ADMIN)", + nsm.device.display() + ) + })?; + Ok(nsm) + } + + pub fn device(&self) -> &Path { + &self.device + } + + /// One raw NSM round trip: CBOR in, CBOR out. + /// + /// This is the whole device interface. Attestation is the same call with a + /// different payload. + pub fn request(&self, cbor: &[u8]) -> Result> { + if cbor.len() > REQUEST_MAX { + bail!( + "NSM request is {} bytes, over the {REQUEST_MAX} limit", + cbor.len() + ); + } + let mut response = vec![0u8; RESPONSE_MAX]; + + let mut raw = NsmRaw { + request: NsmIovec { + addr: cbor.as_ptr() as u64, + len: cbor.len() as u64, + }, + response: NsmIovec { + addr: response.as_mut_ptr() as u64, + // In/out: capacity going in, bytes written coming back. + len: response.len() as u64, + }, + }; + + // SAFETY: the opcode matches `struct nsm_raw`, both iovecs point at + // live allocations that outlive the call, and the lengths describe + // them exactly. + unsafe { + let control = ioctl::Updater::::new(&mut raw); + ioctl::ioctl(self.fd.as_fd(), control) + .with_context(|| format!("NSM ioctl on {}", self.device.display()))?; + } + + let written = raw.response.len as usize; + if written == 0 { + bail!("NSM returned an empty response"); + } + if written > response.len() { + bail!( + "NSM reported {written} bytes into a {} byte buffer", + response.len() + ); + } + response.truncate(written); + Ok(response) + } +} + +impl fmt::Debug for NsmDevice { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("NsmDevice") + .field("device", &self.device) + .finish() + } +} + +impl Nsm for NsmDevice { + fn get_random(&self, buf: &mut [u8]) -> Result<()> { + let mut filled = 0; + while filled < buf.len() { + let response = self.request(&encode_get_random())?; + let bytes = decode_get_random(&response)?; + if bytes.is_empty() { + // Retrying would spin forever against a source that has + // stopped producing. An entropy failure must be loud. + bail!("NSM GetRandom returned no bytes"); + } + let take = bytes.len().min(buf.len() - filled); + buf[filled..filled + take].copy_from_slice(&bytes[..take]); + filled += take; + } + Ok(()) + } + + fn attest(&self, request: &AttestationRequest) -> Result> { + let response = self.request(&encode_attestation(request))?; + decode_attestation(&response) + } + + fn describe_pcr(&self, index: u16) -> Result { + let response = self.request(&encode_pcr_request("DescribePCR", index, None))?; + decode_describe_pcr(&response) + } + + fn extend_pcr(&self, index: u16, data: &[u8]) -> Result> { + let response = self.request(&encode_pcr_request("ExtendPCR", index, Some(data)))?; + decode_extend_pcr(&response) + } + + fn lock_pcr(&self, index: u16) -> Result<()> { + let response = self.request(&encode_pcr_request("LockPCR", index, None))?; + decode_lock_pcr(&response) + } + + fn describe(&self) -> String { + format!("Nitro Security Module ({})", self.device.display()) + } +} + +/// `{"DescribePCR": {"index": n}}`, `{"LockPCR": {"index": n}}`, or +/// `{"ExtendPCR": {"index": n, "data": …}}`. +fn encode_pcr_request(name: &str, index: u16, data: Option<&[u8]>) -> Vec { + let mut fields = vec![( + ciborium::Value::Text("index".into()), + ciborium::Value::Integer(index.into()), + )]; + if let Some(data) = data { + fields.push(( + ciborium::Value::Text("data".into()), + ciborium::Value::Bytes(data.to_vec()), + )); + } + let value = ciborium::Value::Map(vec![( + ciborium::Value::Text(name.to_string()), + ciborium::Value::Map(fields), + )]); + let mut out = Vec::new(); + ciborium::into_writer(&value, &mut out).expect("writing to a Vec cannot fail"); + out +} + +/// Pull the single value out of `{"": {…}}`, surfacing `{"Error": …}`. +fn pcr_response_fields(cbor: &[u8], name: &str) -> Result> { + let value: ciborium::Value = + ciborium::from_reader(cbor).with_context(|| format!("decoding the NSM {name} response"))?; + let outer = value + .as_map() + .with_context(|| format!("NSM response is not a map: {value:?}"))?; + let (key, inner) = outer.first().context("NSM response map is empty")?; + let key = key.as_text().unwrap_or(""); + if key != name { + bail!("NSM answered {key:?} instead of {name}: {inner:?}"); + } + inner + .as_map() + .cloned() + .with_context(|| format!("{name} value is not a map: {inner:?}")) +} + +fn field<'a>( + fields: &'a [(ciborium::Value, ciborium::Value)], + name: &str, +) -> Option<&'a ciborium::Value> { + fields + .iter() + .find(|(k, _)| k.as_text() == Some(name)) + .map(|(_, v)| v) +} + +fn decode_describe_pcr(cbor: &[u8]) -> Result { + let fields = pcr_response_fields(cbor, "DescribePCR")?; + Ok(Pcr { + locked: field(&fields, "lock") + .and_then(|v| match v { + ciborium::Value::Bool(b) => Some(*b), + _ => None, + }) + .context("DescribePCR has no boolean 'lock'")?, + value: field(&fields, "data") + .and_then(|v| v.as_bytes()) + .cloned() + .context("DescribePCR has no 'data'")?, + }) +} + +fn decode_extend_pcr(cbor: &[u8]) -> Result> { + let fields = pcr_response_fields(cbor, "ExtendPCR")?; + field(&fields, "data") + .and_then(|v| v.as_bytes()) + .cloned() + .context("ExtendPCR has no 'data'") +} + +/// A successful `LockPCR` is the bare text string `"LockPCR"`. +/// +/// It has nothing to return, and the device encodes a variant with no fields as +/// its name alone — the shape of the `GetRandom` *request* — so this cannot +/// share [`pcr_response_fields`], which wants a map. A failure is still a map, +/// `{"Error": "ReadOnlyIndex"}`, and its text is what gets reported. +fn decode_lock_pcr(cbor: &[u8]) -> Result<()> { + let value: ciborium::Value = + ciborium::from_reader(cbor).context("decoding the NSM LockPCR response")?; + match &value { + ciborium::Value::Text(name) if name == "LockPCR" => Ok(()), + ciborium::Value::Map(outer) => { + let (key, inner) = outer.first().context("NSM response map is empty")?; + let key = key.as_text().unwrap_or(""); + bail!("NSM answered {key:?} instead of LockPCR: {inner:?}") + } + other => bail!("NSM answered {other:?} instead of LockPCR"), + } +} + +/// `{"Attestation": {"user_data": , ...}}`, with absent fields +/// **omitted rather than set to null**. +/// +/// This is not a style choice. The device parses each present key by pulling a +/// byte string out of it; a CBOR `null` is not a byte string, so a request +/// carrying `"nonce": null` is rejected outright — and the rejection arrives +/// as `InvalidOperation` for the whole request, naming nothing. Omitting the +/// key leaves the field at its default, which is null anyway. +fn encode_attestation(request: &AttestationRequest) -> Vec { + let mut fields = Vec::new(); + let mut push = |name: &str, value: &Option>| { + if let Some(bytes) = value { + fields.push(( + ciborium::Value::Text(name.to_string()), + ciborium::Value::Bytes(bytes.clone()), + )); + } + }; + push("user_data", &request.user_data); + push("nonce", &request.nonce); + push("public_key", &request.public_key); + + let value = ciborium::Value::Map(vec![( + ciborium::Value::Text("Attestation".into()), + ciborium::Value::Map(fields), + )]); + let mut out = Vec::new(); + ciborium::into_writer(&value, &mut out).expect("writing to a Vec cannot fail"); + out +} + +/// Pull the COSE_Sign1 out of `{"Attestation": {"document": }}`. +fn decode_attestation(cbor: &[u8]) -> Result> { + let value: ciborium::Value = + ciborium::from_reader(cbor).context("decoding the NSM Attestation response")?; + + let outer = value + .as_map() + .with_context(|| format!("NSM response is not a map: {value:?}"))?; + let (key, inner) = outer.first().context("NSM response map is empty")?; + let key = key.as_text().unwrap_or(""); + if key != "Attestation" { + // The device reports its own failures as {"Error": "..."} rather than + // by failing the ioctl, so the real text matters here. + bail!("NSM answered {key:?} instead of Attestation: {inner:?}"); + } + + let inner = inner + .as_map() + .with_context(|| format!("Attestation value is not a map: {inner:?}"))?; + for (k, v) in inner { + if k.as_text() == Some("document") { + let doc = v + .as_bytes() + .cloned() + .context("Attestation 'document' field is not a byte string")?; + if doc.is_empty() { + bail!("NSM returned an empty attestation document"); + } + return Ok(doc); + } + } + bail!("Attestation response has no 'document' field") +} + +/// The `GetRandom` request: a bare CBOR text string. +/// +/// The NSM has two request shapes — a plain string for parameterless requests +/// like `GetRandom` and `DescribeNSM`, and a single-entry map for the rest. +fn encode_get_random() -> Vec { + let mut out = Vec::new(); + ciborium::into_writer(&"GetRandom", &mut out).expect("writing to a Vec cannot fail"); + out +} + +/// Pull the bytes out of `{"GetRandom": {"random": }}`. +fn decode_get_random(cbor: &[u8]) -> Result> { + let value: ciborium::Value = + ciborium::from_reader(cbor).context("decoding the NSM GetRandom response")?; + + // The NSM reports its own failures as {"Error": "..."} rather than by + // failing the ioctl, so an unexpected shape deserves the real text. + let outer = value + .as_map() + .with_context(|| format!("NSM response is not a map: {value:?}"))?; + let (key, inner) = outer.first().context("NSM response map is empty")?; + let key = key.as_text().unwrap_or(""); + if key != "GetRandom" { + bail!("NSM answered {key:?} instead of GetRandom: {inner:?}"); + } + + let inner = inner + .as_map() + .with_context(|| format!("GetRandom value is not a map: {inner:?}"))?; + for (k, v) in inner { + if k.as_text() == Some("random") { + return v + .as_bytes() + .cloned() + .context("GetRandom 'random' field is not a byte string"); + } + } + bail!("GetRandom response has no 'random' field") +} + +#[cfg(any(test, feature = "testing"))] +pub mod fake { + use super::*; + use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; + use std::sync::Mutex; + + /// Encodes a GetRandom response the way the device does, so the decoder is + /// tested against the real wire shape rather than against itself. + pub fn encode_response(bytes: &[u8]) -> Vec { + let value = ciborium::Value::Map(vec![( + ciborium::Value::Text("GetRandom".into()), + ciborium::Value::Map(vec![( + ciborium::Value::Text("random".into()), + ciborium::Value::Bytes(bytes.to_vec()), + )]), + )]); + let mut out = Vec::new(); + ciborium::into_writer(&value, &mut out).unwrap(); + out + } + + /// Encodes an Attestation response the way the device does. + pub fn encode_attestation_response(document: &[u8]) -> Vec { + let value = ciborium::Value::Map(vec![( + ciborium::Value::Text("Attestation".into()), + ciborium::Value::Map(vec![( + ciborium::Value::Text("document".into()), + ciborium::Value::Bytes(document.to_vec()), + )]), + )]); + let mut out = Vec::new(); + ciborium::into_writer(&value, &mut out).unwrap(); + out + } + + /// A stand-in device: counts calls, and can be made to return nothing. + #[derive(Debug)] + pub struct FakeNsm { + pub calls: AtomicU64, + pub empty: AtomicBool, + /// Bytes returned per call, mirroring the device's 256-byte chunk. + pub chunk: usize, + /// Document [`Nsm::attest`] hands back, and the last request that + /// asked for one. + /// + /// It is *not* a signed COSE_Sign1 by default — this crate has no + /// signing stack and should not grow one. Tests that need a verifiable + /// document build it in `nitro-attestation` and set it here, so a fake + /// can never accidentally satisfy a real verifier. + pub attestation: Mutex>, + pub last_attestation_request: Mutex>, + /// Real registers, so extend arithmetic and the locked-register + /// refusal can be exercised without a device. + pub pcrs: Mutex>, + } + + impl Default for FakeNsm { + fn default() -> Self { + FakeNsm { + calls: AtomicU64::new(0), + empty: AtomicBool::new(false), + chunk: 256, + attestation: Mutex::new(b"not-a-signed-attestation-document".to_vec()), + last_attestation_request: Mutex::new(None), + // 0–15 locked, as the hypervisor leaves them, with 0–2 carrying + // non-zero image measurements; 16 upwards free and zeroed. + pcrs: Mutex::new( + (0..32) + .map(|i| Pcr { + locked: i < 16, + value: if i < 3 { + vec![0x10 + i as u8; 48] + } else { + PCR_ZERO.to_vec() + }, + }) + .collect(), + ), + } + } + } + + impl FakeNsm { + pub fn new() -> Self { + Self::default() + } + } + + impl Nsm for FakeNsm { + fn get_random(&self, buf: &mut [u8]) -> Result<()> { + if self.empty.load(Ordering::SeqCst) { + bail!("NSM GetRandom returned no bytes"); + } + let mut filled = 0; + while filled < buf.len() { + let n = self.calls.fetch_add(1, Ordering::SeqCst); + let take = self.chunk.min(buf.len() - filled); + // Distinct per call, so a test can tell chunks apart. + for (i, b) in buf[filled..filled + take].iter_mut().enumerate() { + *b = (n as u8) + .wrapping_mul(31) + .wrapping_add(i as u8) + .wrapping_add(1); + } + filled += take; + } + Ok(()) + } + fn attest(&self, request: &AttestationRequest) -> Result> { + if self.empty.load(Ordering::SeqCst) { + bail!("NSM returned an empty attestation document"); + } + *self.last_attestation_request.lock().unwrap() = Some(request.clone()); + Ok(self.attestation.lock().unwrap().clone()) + } + + fn describe_pcr(&self, index: u16) -> Result { + self.pcrs + .lock() + .unwrap() + .get(index as usize) + .cloned() + .context("no such PCR") + } + + fn extend_pcr(&self, index: u16, data: &[u8]) -> Result> { + let mut pcrs = self.pcrs.lock().unwrap(); + let pcr = pcrs.get_mut(index as usize).context("no such PCR")?; + if pcr.locked { + bail!("PCR {index} is read-only"); + } + pcr.value = pcr_extend(&pcr.value, data); + Ok(pcr.value.clone()) + } + + fn lock_pcr(&self, index: u16) -> Result<()> { + let mut pcrs = self.pcrs.lock().unwrap(); + let pcr = pcrs.get_mut(index as usize).context("no such PCR")?; + if pcr.locked { + bail!("PCR {index} is read-only"); + } + pcr.locked = true; + Ok(()) + } + + fn describe(&self) -> String { + "fake NSM".into() + } + } + + #[cfg(test)] + fn cbor(value: ciborium::Value) -> Vec { + let mut out = Vec::new(); + ciborium::into_writer(&value, &mut out).unwrap(); + out + } + + #[test] + fn extending_is_the_documented_arithmetic() { + let nsm = FakeNsm::new(); + let after = nsm.extend_pcr(PCR_GUEST, b"a guest").unwrap(); + assert_eq!(after, pcr_extend(&PCR_ZERO, b"a guest")); + assert_eq!(nsm.describe_pcr(PCR_GUEST).unwrap().value, after); + } + + /// Extension accumulates. Checking `extend(0, x)` is checking that the + /// register was extended exactly once, with `x` — twice, or with anything + /// else first, does not match. + #[test] + fn extending_twice_does_not_look_like_extending_once() { + let nsm = FakeNsm::new(); + nsm.extend_pcr(PCR_GUEST, b"a").unwrap(); + let twice = nsm.extend_pcr(PCR_GUEST, b"b").unwrap(); + assert_ne!(twice, pcr_extend(&PCR_ZERO, b"b")); + } + + /// The fake has to lock what the hypervisor locks, or a test could extend a + /// register no real enclave can touch and pass for the wrong reason. + #[test] + fn the_registers_the_hypervisor_locks_cannot_be_extended() { + let nsm = FakeNsm::new(); + for index in 0..16 { + assert!(nsm.describe_pcr(index).unwrap().locked, "PCR{index}"); + assert!(nsm.extend_pcr(index, b"anything").is_err(), "PCR{index}"); + } + assert!(!nsm.describe_pcr(PCR_GUEST).unwrap().locked); + } + + #[test] + fn a_locked_register_can_be_neither_extended_nor_locked_again() { + let nsm = FakeNsm::new(); + let value = nsm.extend_pcr(PCR_GUEST, b"measured").unwrap(); + nsm.lock_pcr(PCR_GUEST).unwrap(); + + let pcr = nsm.describe_pcr(PCR_GUEST).unwrap(); + assert!(pcr.locked); + assert_eq!(pcr.value, value, "locking must not change the value"); + assert!(nsm.extend_pcr(PCR_GUEST, b"more").is_err()); + assert!(nsm.lock_pcr(PCR_GUEST).is_err()); + assert!(nsm.lock_pcr(32).is_err(), "there is no PCR32"); + } + + /// `{"LockPCR": {"index": 16}}` — a map like the other register requests, + /// and with no `data`, which a lock does not take. + #[test] + fn a_lock_request_names_the_register_and_nothing_else() { + let value: ciborium::Value = + ciborium::from_reader(&encode_pcr_request("LockPCR", PCR_GUEST, None)[..]).unwrap(); + let outer = value.as_map().unwrap(); + assert_eq!(outer.len(), 1); + assert_eq!(outer[0].0.as_text(), Some("LockPCR")); + let fields = outer[0].1.as_map().unwrap(); + assert_eq!(fields.len(), 1, "a lock carries only the index: {fields:?}"); + assert_eq!(fields[0].0.as_text(), Some("index")); + assert_eq!( + fields[0].1.as_integer(), + Some(ciborium::value::Integer::from(PCR_GUEST)) + ); + } + + #[test] + fn a_bare_lock_response_is_success() { + assert!(decode_lock_pcr(&cbor(ciborium::Value::Text("LockPCR".into()))).is_ok()); + } + + /// Locking an already-locked register is the failure worth reading: it + /// means something got to the register first. + #[test] + fn an_error_response_to_a_lock_names_what_came_back() { + let err = decode_lock_pcr(&cbor(ciborium::Value::Map(vec![( + ciborium::Value::Text("Error".into()), + ciborium::Value::Text("ReadOnlyIndex".into()), + )]))) + .unwrap_err(); + let text = format!("{err:#}"); + assert!(text.contains("ReadOnlyIndex"), "{text}"); + } + + #[test] + fn an_answer_to_some_other_request_is_not_a_lock() { + assert!(decode_lock_pcr(&cbor(ciborium::Value::Text("LockPCRs".into()))).is_err()); + assert!(decode_lock_pcr(&[]).is_err()); + } + + /// The device rejects a `null` field outright — `fill_attestation_property` + /// wants a byte string — and answers `InvalidOperation` for the whole + /// request without naming which field was at fault. Omitting absent fields + /// is what keeps that from happening, so it is worth a test. + #[test] + fn absent_attestation_fields_are_omitted_not_nulled() { + let encoded = encode_attestation(&AttestationRequest::with_user_data(vec![1, 2, 3])); + let value: ciborium::Value = ciborium::from_reader(&encoded[..]).unwrap(); + let inner = value.as_map().unwrap()[0].1.as_map().unwrap(); + + let names: Vec<&str> = inner.iter().filter_map(|(k, _)| k.as_text()).collect(); + assert_eq!(names, ["user_data"], "only the supplied field may appear"); + assert!( + !inner + .iter() + .any(|(_, v)| matches!(v, ciborium::Value::Null)), + "a null field makes the device reject the whole request" + ); + } + + #[test] + fn every_supplied_attestation_field_is_carried() { + let request = AttestationRequest { + user_data: Some(vec![0xaa]), + nonce: Some(vec![0xbb]), + public_key: Some(vec![0xcc]), + }; + let value: ciborium::Value = + ciborium::from_reader(&encode_attestation(&request)[..]).unwrap(); + let outer = value.as_map().unwrap(); + assert_eq!(outer[0].0.as_text(), Some("Attestation")); + + let inner = outer[0].1.as_map().unwrap(); + let get = |name: &str| { + inner + .iter() + .find(|(k, _)| k.as_text() == Some(name)) + .map(|(_, v)| v.as_bytes().unwrap().clone()) + }; + assert_eq!(get("user_data"), Some(vec![0xaa])); + assert_eq!(get("nonce"), Some(vec![0xbb])); + assert_eq!(get("public_key"), Some(vec![0xcc])); + } + + #[test] + fn an_attestation_document_round_trips() { + let doc = b"\x84pretend COSE_Sign1".to_vec(); + assert_eq!( + decode_attestation(&encode_attestation_response(&doc)).unwrap(), + doc + ); + } + + /// An empty document is a failure, not an empty success: everything + /// downstream would otherwise try to parse zero bytes and report a + /// confusing parse error instead of "the device gave us nothing". + #[test] + fn an_empty_attestation_document_is_an_error() { + let err = decode_attestation(&encode_attestation_response(&[])).unwrap_err(); + assert!(format!("{err:#}").contains("empty"), "{err:#}"); + } + + #[test] + fn an_error_response_to_attestation_names_what_came_back() { + let value = ciborium::Value::Map(vec![( + ciborium::Value::Text("Error".into()), + ciborium::Value::Text("InvalidOperation".into()), + )]); + let mut cbor = Vec::new(); + ciborium::into_writer(&value, &mut cbor).unwrap(); + + let err = decode_attestation(&cbor).unwrap_err(); + let text = format!("{err:#}"); + assert!(text.contains("Error"), "{text}"); + assert!(text.contains("InvalidOperation"), "{text}"); + } + + #[test] + fn the_request_is_a_bare_cbor_text_string() { + // 0x69 = text string of length 9, then "GetRandom". This is the exact + // shape the device's dispatcher looks for (CBOR_ROOT_TYPE_STRING). + let encoded = encode_get_random(); + assert_eq!(encoded[0], 0x69); + assert_eq!(&encoded[1..], b"GetRandom"); + assert_eq!(encoded.len(), 10); + } + + #[test] + fn a_response_round_trips() { + let bytes: Vec = (0..=255u8).collect(); + assert_eq!(decode_get_random(&encode_response(&bytes)).unwrap(), bytes); + } + + #[test] + fn an_empty_random_field_decodes_to_nothing() { + // Distinct from a malformed response: the caller turns this into a + // hard error rather than looping. + assert!(decode_get_random(&encode_response(&[])).unwrap().is_empty()); + } + + /// The NSM reports its own failures in-band, as a response rather than an + /// ioctl error, so the decoder must surface them rather than say "not a + /// map" and lose the reason. + #[test] + fn an_error_response_names_what_came_back() { + let value = ciborium::Value::Map(vec![( + ciborium::Value::Text("Error".into()), + ciborium::Value::Text("InvalidOperation".into()), + )]); + let mut cbor = Vec::new(); + ciborium::into_writer(&value, &mut cbor).unwrap(); + + let err = format!("{:#}", decode_get_random(&cbor).unwrap_err()); + assert!(err.contains("Error"), "{err}"); + assert!(err.contains("InvalidOperation"), "{err}"); + } + + #[test] + fn malformed_responses_are_rejected() { + assert!(decode_get_random(&[]).is_err()); + assert!(decode_get_random(&[0xff, 0xff, 0xff]).is_err()); + + // Right shape, wrong field. + let value = ciborium::Value::Map(vec![( + ciborium::Value::Text("GetRandom".into()), + ciborium::Value::Map(vec![( + ciborium::Value::Text("entropy".into()), + ciborium::Value::Bytes(vec![1, 2, 3]), + )]), + )]); + let mut cbor = Vec::new(); + ciborium::into_writer(&value, &mut cbor).unwrap(); + assert!(decode_get_random(&cbor).is_err()); + } + + /// The device returns 256 bytes per call, so a larger request must loop + /// and stitch the chunks together rather than truncating. + #[test] + fn a_large_request_loops_over_several_calls() { + let fake = FakeNsm::new(); + let mut buf = vec![0u8; 1000]; + fake.get_random(&mut buf).unwrap(); + + assert_eq!(fake.calls.load(Ordering::SeqCst), 4, "1000 bytes / 256"); + assert!(buf.iter().any(|&b| b != 0)); + // Chunks differ from one another, so nothing was copied twice. + assert_ne!(&buf[0..256], &buf[256..512]); + } + + #[test] + fn a_source_that_stops_producing_is_an_error_not_a_spin() { + let fake = FakeNsm::new(); + fake.empty.store(true, Ordering::SeqCst); + assert!(fake.get_random(&mut [0u8; 16]).is_err()); + } + + #[test] + fn opening_a_missing_device_names_it() { + let err = NsmDevice::open("/dev/definitely-not-nsm").unwrap_err(); + assert!(format!("{err:#}").contains("/dev/definitely-not-nsm")); + } +} diff --git a/crates/s3fs-core/Cargo.toml b/crates/s3fs-core/Cargo.toml index b93a742..1bddf53 100644 --- a/crates/s3fs-core/Cargo.toml +++ b/crates/s3fs-core/Cargo.toml @@ -26,6 +26,9 @@ siphasher = { workspace = true } hashlink = { workspace = true } bitvec = { workspace = true } percent-encoding = { workspace = true } +blake3 = { workspace = true } +zeroize = { workspace = true } +aws-lc-rs = { workspace = true } aws-config = { workspace = true, optional = true, features = ["behavior-version-latest", "rt-tokio"] } aws-sdk-s3 = { workspace = true, optional = true, features = ["rt-tokio", "rustls", "behavior-version-latest"] } diff --git a/crates/s3fs-core/benches/fs_hot_paths.rs b/crates/s3fs-core/benches/fs_hot_paths.rs index 44d245e..9f0e65d 100644 --- a/crates/s3fs-core/benches/fs_hot_paths.rs +++ b/crates/s3fs-core/benches/fs_hot_paths.rs @@ -1,119 +1,95 @@ -//! Benchmarks for the Fs hot paths against `MemoryBackend`. The numbers -//! are useful for relative comparison (regressions, the impact of design -//! changes); they're not representative of real S3 latency since -//! MemoryBackend is in-process. +//! Hot-path benchmarks against the in-memory backend. //! -//! Run with `cargo bench -p s3fs-core`. Add `--bench fs_hot_paths` to -//! filter. To compare two revisions, use `criterion`'s baselines: -//! `cargo bench -p s3fs-core -- --save-baseline before`, -//! then `cargo bench -p s3fs-core -- --baseline before`. +//! Numbers here are relative only — there is no network, so what they measure +//! is the engine's own cost: encryption, hashing, the copy-on-write rebuild, +//! and the commit protocol. That is exactly the part worth watching, because +//! against real S3 the round trips dominate everything else. +//! +//! ```bash +//! cargo bench -p s3fs-core +//! ``` use std::sync::Arc; -use bytes::Bytes; -use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; +use criterion::{criterion_group, criterion_main, BatchSize, BenchmarkId, Criterion, Throughput}; +use tokio::runtime::Runtime; use s3fs_core::backend::memory::MemoryBackend; -use s3fs_core::backend::{Backend, PutBlobInput}; -use s3fs_core::config::PartSchedule; -use s3fs_core::{Config, Fs, OpenFlags}; - -fn build_fs(part_size: u64) -> (Arc, Arc) { - let backend = Arc::new(MemoryBackend::new()); - let cfg = Config::builder() - .part_schedule(PartSchedule { - tiers: vec![(part_size, 1000)], - }) - .single_part_threshold(part_size) - .max_parallel_parts(8) - .max_parallel_copy(8) - .max_merge_copy_bytes(128 * 1024 * 1024) - .memory_limit_bytes(64 * 1024 * 1024) +use s3fs_core::backend::Backend; +use s3fs_core::{Config, Fs, MasterSecret, OpenFlags}; + +fn runtime() -> Runtime { + tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("tokio runtime") +} + +async fn fresh_fs(record_size: usize) -> Arc { + let config = Config::builder() + .record_size(record_size) + .root_retention(None) .build(); - let fs = Fs::new(backend.clone() as Arc, Arc::new(cfg)); - (backend, fs) + Fs::create( + Arc::new(MemoryBackend::new()) as Arc, + Arc::new(MemoryBackend::new()) as Arc, + &MasterSecret::from_bytes([1u8; 32]), + [0u8; 16], + Arc::new(config), + ) + .await + .expect("creating the filesystem") } -/// Open a file, write `payload`, sync, close. End-to-end small-file path. -fn bench_small_file_write_sync(c: &mut Criterion) { - let rt = tokio::runtime::Builder::new_current_thread() - .build() - .unwrap(); - let mut group = c.benchmark_group("small_file_write_sync"); +/// Write and commit a whole file: the full path through encryption, the +/// indirect-tree rebuild, slab packing, and a signed root record. +fn write_and_commit(c: &mut Criterion) { + let rt = runtime(); + let mut group = c.benchmark_group("write_and_commit"); + for size in [1024usize, 64 * 1024, 1024 * 1024] { group.throughput(Throughput::Bytes(size as u64)); group.bench_with_input(BenchmarkId::from_parameter(size), &size, |b, &size| { - let payload = vec![0xAB_u8; size]; - b.to_async(&rt).iter(|| { - let payload = payload.clone(); - async move { - let (_backend, fs) = build_fs(5 * 1024 * 1024); - let h = fs - .open( - "k", - OpenFlags { - read: true, - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - fs.pwrite(&h, 0, &payload).await.unwrap(); - fs.sync(&h).await.unwrap(); + let data = vec![0xabu8; size]; + b.to_async(&rt).iter_batched( + || data.clone(), + |data| async move { + let fs = fresh_fs(128 * 1024).await; + let h = fs.open("/bench", OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&h, 0, &data).await.unwrap(); fs.close(&h).await.unwrap(); - black_box(()) - } - }); + }, + BatchSize::SmallInput, + ); }); } group.finish(); } -/// Pre-populate a 30 MiB file and `pread` random offsets via the buffer -/// pool. Measures cache-miss + cache-hit read paths. -fn bench_pread_buffered(c: &mut Criterion) { - let rt = tokio::runtime::Builder::new_current_thread() - .build() - .unwrap(); - let mut group = c.benchmark_group("pread_buffered"); - for read_size in [4 * 1024usize, 64 * 1024, 1024 * 1024] { - group.throughput(Throughput::Bytes(read_size as u64)); +/// Commit cost as a function of how many records the transaction group dirties. +/// +/// The interesting shape: slab packing means the number of PUTs stays flat, so +/// this should scale with bytes rather than with block count. +fn commit_by_dirty_blocks(c: &mut Criterion) { + let rt = runtime(); + let mut group = c.benchmark_group("commit_by_dirty_blocks"); + let record = 4096usize; + + for blocks in [1usize, 16, 256] { + group.throughput(Throughput::Elements(blocks as u64)); group.bench_with_input( - BenchmarkId::from_parameter(read_size), - &read_size, - |b, &read_size| { - let setup = || { - let part_size = 5 * 1024 * 1024u64; - let (backend, fs) = build_fs(part_size); - let body = Bytes::from(vec![0u8; 30 * 1024 * 1024]); - let backend = backend.clone(); - let fs = fs.clone(); - rt.block_on(async move { - backend - .put_blob(PutBlobInput { - key: "big".into(), - body, - metadata: Default::default(), - content_type: None, - }) - .await - .unwrap(); - let h = fs.open("big", OpenFlags::read_only()).await.unwrap(); - (fs, h) - }) - }; - let (fs, h) = setup(); - b.to_async(&rt).iter(|| { - let fs = fs.clone(); - let h = h.clone(); - async move { - // Read at a moving offset to vary pool hits/misses. - let off = (fastrand::u64(..) % (29 * 1024 * 1024)) as u64; - let r = fs.pread(&h, off, read_size).await.unwrap(); - black_box(r); + BenchmarkId::from_parameter(blocks), + &blocks, + |b, &blocks| { + b.to_async(&rt).iter(|| async move { + let fs = fresh_fs(record).await; + let h = fs.open("/bench", OpenFlags::create_new()).await.unwrap(); + // One byte per record, so every record is dirtied but the + // volume of data stays small. + for i in 0..blocks { + fs.pwrite(&h, (i * record) as u64, b"x").await.unwrap(); } + fs.close(&h).await.unwrap(); }); }, ); @@ -121,50 +97,70 @@ fn bench_pread_buffered(c: &mut Criterion) { group.finish(); } -/// In-place edit of a multi-part file: 30 MiB existing object, modify -/// `dirty_kb` at part 1, sync. Exercises the GeeseFS-parity MPU + -/// UploadPartCopy materialisation path and the parallel-upload flusher. -fn bench_inplace_subpart_edit(c: &mut Criterion) { - let rt = tokio::runtime::Builder::new_current_thread() - .build() - .unwrap(); - let mut group = c.benchmark_group("inplace_subpart_edit"); - for dirty_kb in [4_usize, 64, 1024] { - group.throughput(Throughput::Bytes((dirty_kb * 1024) as u64)); +/// Random reads over a committed file, warm cache. Measures verification cost: +/// a BLAKE3 pass plus an AEAD open per block, plus the indirect-tree walk. +fn random_reads(c: &mut Criterion) { + let rt = runtime(); + let file_size = 8 * 1024 * 1024usize; + + let fs = rt.block_on(async { + let fs = fresh_fs(128 * 1024).await; + let h = fs.open("/bench", OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&h, 0, &vec![0x5au8; file_size]).await.unwrap(); + fs.close(&h).await.unwrap(); + fs + }); + + let mut group = c.benchmark_group("random_read"); + for len in [4096usize, 64 * 1024] { + group.throughput(Throughput::Bytes(len as u64)); + group.bench_with_input(BenchmarkId::from_parameter(len), &len, |b, &len| { + let fs = fs.clone(); + b.to_async(&rt).iter(|| { + let fs = fs.clone(); + async move { + let h = fs.open("/bench", OpenFlags::read_only()).await.unwrap(); + let offset = fastrand::u64(0..(file_size - len) as u64); + fs.pread(&h, offset, len).await.unwrap(); + fs.close(&h).await.unwrap(); + } + }); + }); + } + group.finish(); +} + +/// Directory lookup as the directory grows. The separator index means this +/// should stay roughly flat rather than degrading with entry count. +fn directory_lookup(c: &mut Criterion) { + let rt = runtime(); + let mut group = c.benchmark_group("directory_lookup"); + + for entries in [10usize, 1_000, 10_000] { + let fs = rt.block_on(async { + let fs = fresh_fs(4096).await; + let root = fs.root(); + for i in 0..entries { + let h = fs + .open(&format!("/entry-{i:06}"), OpenFlags::create_new()) + .await + .unwrap(); + fs.close(&h).await.unwrap(); + } + let _ = root; + fs + }); + group.bench_with_input( - BenchmarkId::from_parameter(dirty_kb), - &dirty_kb, - |b, &dirty_kb| { + BenchmarkId::from_parameter(entries), + &entries, + |b, &entries| { + let fs = fs.clone(); b.to_async(&rt).iter(|| { - let dirty = vec![0xAB_u8; dirty_kb * 1024]; + let fs = fs.clone(); async move { - let part_size = 5 * 1024 * 1024u64; - let (backend, fs) = build_fs(part_size); - let body = Bytes::from(vec![0u8; 30 * 1024 * 1024]); - backend - .put_blob(PutBlobInput { - key: "k".into(), - body, - metadata: Default::default(), - content_type: None, - }) - .await - .unwrap(); - let h = fs - .open( - "k", - OpenFlags { - read: true, - write: true, - ..Default::default() - }, - ) - .await - .unwrap(); - fs.pwrite(&h, part_size, &dirty).await.unwrap(); - fs.sync(&h).await.unwrap(); - fs.close(&h).await.unwrap(); - black_box(()) + let name = format!("entry-{:06}", fastrand::usize(0..entries)); + fs.lookup_at(&fs.root(), &name).await.unwrap(); } }); }, @@ -175,8 +171,9 @@ fn bench_inplace_subpart_edit(c: &mut Criterion) { criterion_group!( benches, - bench_small_file_write_sync, - bench_pread_buffered, - bench_inplace_subpart_edit, + write_and_commit, + commit_by_dirty_blocks, + random_reads, + directory_lookup ); criterion_main!(benches); diff --git a/crates/s3fs-core/src/backend/aws.rs b/crates/s3fs-core/src/backend/aws.rs index c29e0e8..255061c 100644 --- a/crates/s3fs-core/src/backend/aws.rs +++ b/crates/s3fs-core/src/backend/aws.rs @@ -19,7 +19,7 @@ use aws_sdk_s3::operation::head_object::HeadObjectError; use aws_sdk_s3::primitives::ByteStream; use aws_sdk_s3::types::{ CompletedMultipartUpload, CompletedPart as SdkCompletedPart, Delete, MetadataDirective, - ObjectIdentifier, + ObjectIdentifier, ObjectLockMode as SdkObjectLockMode, }; use aws_sdk_s3::Client; use aws_smithy_runtime_api::client::result::SdkError; @@ -28,10 +28,30 @@ use bytes::Bytes; use super::{ Backend, BlobItem, BlobMeta, Capabilities, CompletedPart, CopyBlobInput, GetBlobOutput, - ListBlobsInput, ListBlobsOutput, MultipartId, PartUploadOutput, PutBlobInput, + ListBlobsInput, ListBlobsOutput, MultipartId, ObjectLock, ObjectLockMode, PartUploadOutput, + PutBlobInput, }; use crate::errors::{FsError, FsResult}; +/// Apply Object Lock retention headers to a `PutObject` request. +/// +/// Note the bucket must have been created with Object Lock enabled +/// (`ObjectLockEnabledForBucket`); S3 rejects these headers otherwise. That +/// failure surfaces as `InvalidRequest` from `send()` rather than being +/// swallowed here, which is the behaviour we want — a retention we asked for +/// and didn't get must be loud. +fn apply_object_lock( + req: aws_sdk_s3::operation::put_object::builders::PutObjectFluentBuilder, + lock: &ObjectLock, +) -> aws_sdk_s3::operation::put_object::builders::PutObjectFluentBuilder { + let mode = match lock.mode { + ObjectLockMode::Governance => SdkObjectLockMode::Governance, + ObjectLockMode::Compliance => SdkObjectLockMode::Compliance, + }; + req.object_lock_mode(mode) + .object_lock_retain_until_date(aws_smithy_types::DateTime::from(lock.retain_until)) +} + /// Construction parameters for [`AwsS3Backend`]. #[derive(Debug, Clone)] pub struct AwsS3BackendConfig { @@ -73,6 +93,7 @@ impl AwsS3BackendConfig { pub struct AwsS3Backend { bucket: String, client: Client, + config: AwsS3BackendConfig, } impl AwsS3Backend { @@ -98,7 +119,26 @@ impl AwsS3Backend { let mut conf_builder = aws_sdk_s3::Config::builder() .behavior_version(aws_sdk_s3::config::BehaviorVersion::latest()) .region(Region::new(config.region.clone())) - .force_path_style(config.force_path_style); + .force_path_style(config.force_path_style) + // `request_timeout` was stored on this config and never applied. + // The SDK's defaults give a connect timeout and **no operation + // timeout**, so a read that stalls after the connection is + // established hangs forever — and inside an enclave that hangs + // whatever called it: a mount, a commit, or a guest holding its + // tenant's lock open across a response body. Nothing above can + // rescue it either, because a task parked in a host future + // executes no wasm and so is invisible to the epoch watchdog. + // + // Per attempt, with the SDK's retries on top, which is what this + // field's own documentation describes. The overall ceiling is three + // times that — the standard retry policy's attempt count — so a + // pathological endpoint cannot stretch one operation without limit. + .timeout_config( + aws_smithy_types::timeout::TimeoutConfig::builder() + .operation_attempt_timeout(config.request_timeout) + .operation_timeout(config.request_timeout * 3) + .build(), + ); if let Some(ep) = &config.endpoint { conf_builder = conf_builder.endpoint_url(ep); } @@ -117,8 +157,9 @@ impl AwsS3Backend { } let client = Client::from_conf(conf_builder.build()); Ok(Self { - bucket: config.bucket, + bucket: config.bucket.clone(), client, + config, }) } @@ -126,6 +167,12 @@ impl AwsS3Backend { &self.bucket } + /// The configuration this backend was built from, so a caller can derive + /// a second backend against another bucket on the same endpoint. + pub fn config(&self) -> &AwsS3BackendConfig { + &self.config + } + pub fn client(&self) -> &Client { &self.client } @@ -139,6 +186,7 @@ impl Backend for AwsS3Backend { upload_part_copy: true, batch_delete: true, metadata_in_listings: false, + object_lock: true, } } @@ -217,6 +265,99 @@ impl Backend for AwsS3Backend { }) } + async fn get_retained_blob(&self, key: &str) -> FsResult { + // `ListObjectVersions` still reports a version that a delete marker is + // hiding, so this is what tells "never written" from "hidden". The + // prefix is not an exact match, hence the `Key == key` filter. + // + // Every page, not the first: delete markers count against the page + // size and anyone with DeleteObject can stack a thousand of them on a + // key, pushing the retained version onto a later page. Stopping early + // would report it absent. + let mut versions = Vec::new(); + let mut key_marker = None; + let mut version_id_marker = None; + loop { + let listed = self + .client + .list_object_versions() + .bucket(&self.bucket) + .prefix(key) + .set_key_marker(key_marker.take()) + .set_version_id_marker(version_id_marker.take()) + .send() + .await + .map_err(|e| map_sdk_error("ListObjectVersions", e))?; + versions.extend( + listed + .versions + .unwrap_or_default() + .into_iter() + .filter(|v| v.key.as_deref() == Some(key)), + ); + if !listed.is_truncated.unwrap_or(false) { + break; + } + key_marker = listed.next_key_marker; + version_id_marker = listed.next_version_id_marker; + if key_marker.is_none() { + return Err(FsError::Io( + "ListObjectVersions: truncated without a continuation marker".into(), + )); + } + } + + // The oldest, so the version the conditional PUT created wins over + // anything layered on later. By position, not `last_modified`: S3 + // lists a key's versions newest first, page after page, while two + // versions written close together can share a timestamp — and sorting + // on a tie would leave the newer one in front. + let version_id = versions + .into_iter() + .rev() + .find_map(|v| v.version_id) + .ok_or(FsError::NotFound)?; + + let resp = self + .client + .get_object() + .bucket(&self.bucket) + .key(key) + .version_id(&version_id) + .send() + .await + .map_err(|e| map_sdk_error("GetObject(versionId)", e))?; + + let e_tag = resp.e_tag.clone().unwrap_or_default(); + let last_modified = resp + .last_modified + .as_ref() + .map(dt_to_systemtime) + .unwrap_or(std::time::UNIX_EPOCH); + let content_type = resp.content_type.clone(); + let metadata = resp.metadata.clone().unwrap_or_default(); + let body = resp + .body + .collect() + .await + .map_err(|e| FsError::Io(format!("GetObject(versionId) body: {e}")))? + .into_bytes(); + let size = body.len() as u64; + + Ok(GetBlobOutput { + meta: BlobMeta { + key: key.to_string(), + e_tag, + size, + last_modified, + content_type, + metadata, + is_dir_marker: key.ends_with('/'), + }, + body, + }) + } + async fn put_blob(&self, input: PutBlobInput) -> FsResult { let key = input.key.clone(); let body = ByteStream::from(input.body.to_vec()); @@ -234,6 +375,9 @@ impl Backend for AwsS3Backend { if let Some(ct) = &input.content_type { req = req.content_type(ct); } + if let Some(lock) = &input.object_lock { + req = apply_object_lock(req, lock); + } let resp = req .send() .await @@ -267,6 +411,9 @@ impl Backend for AwsS3Backend { if let Some(ct) = &input.content_type { req = req.content_type(ct); } + if let Some(lock) = &input.object_lock { + req = apply_object_lock(req, lock); + } let resp = req.send().await.map_err(|e| { // S3 returns 412 PreconditionFailed when If-None-Match: * fires. match service_code(&e) { @@ -434,6 +581,15 @@ impl Backend for AwsS3Backend { if let Some(ct) = &input.content_type { req = req.content_type(ct); } + if let Some(lock) = &input.object_lock { + let mode = match lock.mode { + ObjectLockMode::Governance => SdkObjectLockMode::Governance, + ObjectLockMode::Compliance => SdkObjectLockMode::Compliance, + }; + req = req + .object_lock_mode(mode) + .object_lock_retain_until_date(aws_smithy_types::DateTime::from(lock.retain_until)); + } let resp = req .send() .await @@ -598,3 +754,102 @@ where _ => FsError::Io(format!("{e:?}")), } } + +#[cfg(test)] +mod timeout_tests { + use super::*; + + /// The bug this guards: the field existed, was defaulted, was threaded + /// through two layers of config — and was never handed to the SDK. Asserted + /// on the built client rather than on our own struct, because our struct + /// held the right value the whole time it was being ignored. + #[tokio::test] + async fn the_request_timeout_reaches_the_client() { + let mut config = AwsS3BackendConfig::new("bucket", "us-east-1"); + config.request_timeout = Duration::from_secs(7); + config.endpoint = Some("http://127.0.0.1:1".into()); + + let backend = AwsS3Backend::connect_unchecked(config) + .await + .expect("building a client does no I/O"); + let timeouts = backend + .client() + .config() + .timeout_config() + .expect("the SDK was given no timeout configuration at all"); + + assert_eq!( + timeouts.operation_attempt_timeout(), + Some(Duration::from_secs(7)), + "one attempt is unbounded" + ); + assert_eq!( + timeouts.operation_timeout(), + Some(Duration::from_secs(21)), + "the operation as a whole is unbounded" + ); + } +} + +#[cfg(test)] +mod retained_tests { + use super::*; + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + + /// Two versions of one key stamped in the same second, listed as S3 lists + /// them: newest first. Sorting on the timestamp left that order alone, so + /// the "retained" record was the one a second writer layered on after + /// hiding the first — and root publication reads it back to decide who + /// won a sequence. + #[tokio::test] + async fn the_retained_version_is_the_first_written_even_on_a_timestamp_tie() { + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let addr = listener.local_addr().unwrap(); + tokio::spawn(async move { + loop { + let (mut socket, _) = listener.accept().await.unwrap(); + let mut request = vec![0; 8192]; + let n = socket.read(&mut request).await.unwrap(); + let request = String::from_utf8_lossy(&request[..n]).into_owned(); + let line = request.lines().next().unwrap_or_default(); + let body = if line.contains("versionId=v-first") { + "first".to_string() + } else if line.contains("versionId=") { + "later".to_string() + } else { + let version = |id: &str, latest: bool| { + format!( + "roots/1{id}\ + {latest}\ + 2026-01-01T00:00:00.000Z" + ) + }; + format!( + "\ + \ + rootsroots/1false\ + {}{}", + version("v-later", true), + version("v-first", false), + ) + }; + let response = format!( + "HTTP/1.1 200 OK\r\nContent-Type: application/xml\r\n\ + Content-Length: {}\r\nConnection: close\r\n\r\n{body}", + body.len() + ); + socket.write_all(response.as_bytes()).await.unwrap(); + } + }); + + let mut config = AwsS3BackendConfig::new("roots", "us-east-1"); + config.endpoint = Some(format!("http://{addr}")); + config.force_path_style = true; + config.access_key_id = Some("test".into()); + config.secret_access_key = Some("test".into()); + let backend = AwsS3Backend::connect_unchecked(config).await.unwrap(); + + let got = backend.get_retained_blob("roots/1").await.unwrap(); + assert_eq!(got.body.as_ref(), b"first"); + } +} diff --git a/crates/s3fs-core/src/backend/memory.rs b/crates/s3fs-core/src/backend/memory.rs index 0a3400e..e8dbe36 100644 --- a/crates/s3fs-core/src/backend/memory.rs +++ b/crates/s3fs-core/src/backend/memory.rs @@ -3,6 +3,26 @@ //! conditional create, multipart uploads with `UploadPartCopy`, last-write-wins. //! //! Purely a test fake. Not durable. Not thread-safe across processes. +//! +//! ## What it models of Object Lock, and what it does not +//! +//! It used to describe itself as a non-versioned bucket, which cannot exist: +//! S3 requires versioning for Object Lock. The consequence was worse than +//! inaccuracy — the fake was *stronger* than the real thing, so a test could +//! prove a retained object undeletable when in reality it is only +//! un-destroyable. +//! +//! Now modelled, because it defeats a check that has no signature to fall back +//! on: **a delete marker**. `delete_blob` on a retained object succeeds and +//! hides it from `get_blob`, `head_blob` and `list_blobs`, exactly as S3 does, +//! while the bytes remain reachable through [`Backend::get_retained_blob`]. +//! +//! Still *not* modelled, deliberately: a plain `put_blob` over a retained +//! object, which real S3 accepts as a new version. Every such object here is +//! content-verified — a root record by its Ed25519 signature, a receipt by the +//! NSM's — so substituted bytes are refused a layer up. It is the *absence* +//! that has nothing to verify, which is why that half is modelled and this one +//! is named instead. use std::collections::HashMap; use std::ops::Range; @@ -18,7 +38,7 @@ use std::hash::Hasher; use super::{ Backend, BlobItem, BlobMeta, Capabilities, CompletedPart, CopyBlobInput, GetBlobOutput, - ListBlobsInput, ListBlobsOutput, MultipartId, PartUploadOutput, PutBlobInput, + ListBlobsInput, ListBlobsOutput, MultipartId, ObjectLock, PartUploadOutput, PutBlobInput, }; use crate::errors::{FsError, FsResult}; @@ -29,6 +49,20 @@ struct StoredBlob { last_modified: SystemTime, content_type: Option, metadata: HashMap, + /// Retention stamped at PUT time, if any. + object_lock: Option, +} + +impl StoredBlob { + /// `true` if retention is still in force, i.e. S3 would refuse to delete + /// or overwrite this object. + /// + /// Governance mode is modelled as equally binding: this fake has no notion + /// of IAM principals, so there is nobody here who could hold + /// `s3:BypassGovernanceRetention`. + fn is_retained(&self, now: SystemTime) -> bool { + self.object_lock.is_some_and(|lock| now < lock.retain_until) + } } #[derive(Debug, Clone)] @@ -36,6 +70,7 @@ struct ActiveMpu { key: String, metadata: HashMap, content_type: Option, + object_lock: Option, parts: HashMap, } @@ -48,15 +83,71 @@ struct MpuPart { #[derive(Debug, Default)] struct State { objects: HashMap, + /// Keys with a delete marker on top: the object is still stored, and every + /// ordinary read must behave as though it is not there. + delete_markers: std::collections::HashSet, + /// The version Object Lock pinned, for keys that have since been written + /// over. A new PUT stacks a version on top and leaves the retained one + /// where it was — it is the *current* pointer that moves, which is the + /// distinction the whole delete-marker problem turns on. + retained: HashMap, mpus: HashMap, } +impl State { + /// What an ordinary `GetObject` would see. + fn current(&self, key: &str) -> Option<&StoredBlob> { + if self.delete_markers.contains(key) { + return None; + } + self.objects.get(key) + } + + /// What `ListObjectVersions` plus a versioned `GetObject` would find: the + /// pinned version if one has been written over, otherwise whatever is + /// stored — marker or no marker. + fn retained(&self, key: &str) -> Option<&StoredBlob> { + self.retained.get(key).or_else(|| self.objects.get(key)) + } + + /// Called before anything replaces `objects[key]`. Retention cannot be + /// undone by writing, so the pinned version is set aside rather than lost. + fn pin_before_overwrite(&mut self, key: &str, now: SystemTime) { + if self.retained.contains_key(key) { + return; + } + if let Some(b) = self.objects.get(key) { + if b.is_retained(now) { + self.retained.insert(key.to_string(), b.clone()); + } + } + } +} + +/// One key's worth of `DeleteObject`. +/// +/// Retained: a delete marker goes on top and the object stays, which is what S3 +/// does and is not at all the same as refusing. Unretained: the object really +/// goes. Idempotent either way — S3 answers success for a key that was never +/// there. +fn delete_one(state: &mut State, key: &str, now: SystemTime) { + if state.objects.get(key).is_some_and(|b| b.is_retained(now)) { + state.delete_markers.insert(key.to_string()); + return; + } + state.objects.remove(key); + state.delete_markers.remove(key); +} + /// In-memory backend. #[derive(Debug, Clone)] pub struct MemoryBackend { state: Arc>, next_mpu_id: Arc, next_etag: Arc, + /// Test-helper: an error the next PUT, conditional or not, returns + /// instead of storing. + fail_next_put: Arc>>, } impl Default for MemoryBackend { @@ -71,9 +162,15 @@ impl MemoryBackend { state: Arc::new(RwLock::new(State::default())), next_mpu_id: Arc::new(AtomicU64::new(1)), next_etag: Arc::new(AtomicU64::new(1)), + fail_next_put: Arc::new(RwLock::new(None)), } } + /// Test-helper: make the next PUT fail with `e`, storing nothing. + pub fn fail_next_put(&self, e: FsError) { + *self.fail_next_put.write() = Some(e); + } + /// Test-helper: count of committed objects. pub fn object_count(&self) -> usize { self.state.read().objects.len() @@ -122,18 +219,19 @@ impl Backend for MemoryBackend { upload_part_copy: true, batch_delete: true, metadata_in_listings: false, // mirror standard S3 + object_lock: true, } } async fn head_blob(&self, key: &str) -> FsResult { let g = self.state.read(); - let b = g.objects.get(key).ok_or(FsError::NotFound)?; + let b = g.current(key).ok_or(FsError::NotFound)?; Ok(self.meta_from_stored(key, b)) } async fn get_blob(&self, key: &str, range: Option>) -> FsResult { let g = self.state.read(); - let b = g.objects.get(key).ok_or(FsError::NotFound)?; + let b = g.current(key).ok_or(FsError::NotFound)?; let meta = self.meta_from_stored(key, b); let body = match range { None => b.body.clone(), @@ -149,7 +247,20 @@ impl Backend for MemoryBackend { Ok(GetBlobOutput { meta, body }) } + /// Reads straight through a delete marker, which is the whole point. + async fn get_retained_blob(&self, key: &str) -> FsResult { + let g = self.state.read(); + let b = g.retained(key).ok_or(FsError::NotFound)?; + Ok(GetBlobOutput { + meta: self.meta_from_stored(key, b), + body: b.body.clone(), + }) + } + async fn put_blob(&self, input: PutBlobInput) -> FsResult { + if let Some(e) = self.fail_next_put.write().take() { + return Err(e); + } let e_tag = self.make_etag(&input.body); let now = SystemTime::now(); let stored = StoredBlob { @@ -158,26 +269,43 @@ impl Backend for MemoryBackend { last_modified: now, content_type: input.content_type, metadata: input.metadata, + object_lock: input.object_lock, }; let key = input.key; let mut g = self.state.write(); + // This fake models a non-versioned bucket, so an overwrite replaces + // the retained version rather than adding a new one. Refuse it. + if g.objects.get(&key).is_some_and(|b| b.is_retained(now)) { + return Err(FsError::AccessDenied); + } g.objects.insert(key.clone(), stored.clone()); Ok(self.meta_from_stored(&key, &stored)) } async fn put_blob_if_not_exists(&self, input: PutBlobInput) -> FsResult { + if let Some(e) = self.fail_next_put.write().take() { + return Err(e); + } let mut g = self.state.write(); - if g.objects.contains_key(&input.key) { + // `If-None-Match: *` tests the *current* version, and a delete marker + // is a current version that is not an object — so a hidden key accepts + // a conditional create, and the enclave that makes one believes it is + // the first ever to write here. That is the attack this fake now + // permits so it can be tested. + if g.current(&input.key).is_some() { return Err(FsError::AlreadyExists); } - let e_tag = self.make_etag(&input.body); let now = SystemTime::now(); + g.pin_before_overwrite(&input.key, now); + g.delete_markers.remove(&input.key); + let e_tag = self.make_etag(&input.body); let stored = StoredBlob { body: input.body, e_tag, last_modified: now, content_type: input.content_type, metadata: input.metadata, + object_lock: input.object_lock, }; let key = input.key; g.objects.insert(key.clone(), stored.clone()); @@ -185,25 +313,30 @@ impl Backend for MemoryBackend { } async fn delete_blob(&self, key: &str) -> FsResult<()> { - // S3 `DeleteObject` is idempotent — non-existent key returns success. - self.state.write().objects.remove(key); + let now = SystemTime::now(); + let mut g = self.state.write(); + delete_one(&mut g, key, now); Ok(()) } async fn delete_blobs(&self, keys: &[String]) -> FsResult<()> { + let now = SystemTime::now(); let mut g = self.state.write(); for k in keys { - g.objects.remove(k); + delete_one(&mut g, k, now); } Ok(()) } async fn list_blobs(&self, input: ListBlobsInput<'_>) -> FsResult { let g = self.state.read(); + // `ListObjectsV2` lists current versions, so a marked key is absent + // from it too. `ListObjectVersions` is the call that still sees them, + // and `get_retained_blob` is this fake's stand-in for it. let mut keys: Vec = g .objects .keys() - .filter(|k| k.starts_with(input.prefix)) + .filter(|k| k.starts_with(input.prefix) && !g.delete_markers.contains(*k)) .cloned() .collect(); keys.sort(); @@ -274,15 +407,26 @@ impl Backend for MemoryBackend { .ok_or(FsError::NotFound)? .clone(); + let now = SystemTime::now(); + if g.objects + .get(&input.destination_key) + .is_some_and(|b| b.is_retained(now)) + { + return Err(FsError::AccessDenied); + } + let metadata = input.replace_metadata.unwrap_or(src.metadata); let content_type = input.replace_content_type.or(src.content_type); let e_tag = self.make_etag(&src.body); let stored = StoredBlob { body: src.body, e_tag, - last_modified: SystemTime::now(), + last_modified: now, content_type, metadata, + // S3 does not carry retention across a copy unless the request + // asks for it, and `CopyBlobInput` has no way to ask. + object_lock: None, }; g.objects .insert(input.destination_key.clone(), stored.clone()); @@ -292,7 +436,9 @@ impl Backend for MemoryBackend { async fn multipart_begin(&self, input: PutBlobInput) -> FsResult { let n = self.next_mpu_id.fetch_add(1, Ordering::Relaxed); let id = MultipartId(format!("mpu-{n}")); + let object_lock = input.object_lock; let mpu = ActiveMpu { + object_lock, key: input.key, metadata: input.metadata, content_type: input.content_type, @@ -402,12 +548,17 @@ impl Backend for MemoryBackend { let body = Bytes::from(body_acc); let e_tag = self.make_etag(&body); + let now = SystemTime::now(); + if g.objects.get(key).is_some_and(|b| b.is_retained(now)) { + return Err(FsError::AccessDenied); + } let stored = StoredBlob { body, e_tag, - last_modified: SystemTime::now(), + last_modified: now, content_type: mpu.content_type, metadata: mpu.metadata, + object_lock: mpu.object_lock, }; g.objects.insert(key.to_string(), stored.clone()); Ok(self.meta_from_stored(key, &stored)) @@ -439,6 +590,7 @@ mod tests { body: Bytes::from_static(b"hello"), metadata: Default::default(), content_type: Some("text/plain".into()), + object_lock: None, }) .await .unwrap(); @@ -466,6 +618,7 @@ mod tests { body: Bytes::from_static(b"abcdef"), metadata: Default::default(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -481,6 +634,7 @@ mod tests { body: Bytes::from_static(b"v"), metadata: Default::default(), content_type: None, + object_lock: None, }; b.put_blob_if_not_exists(body()).await.unwrap(); assert!(matches!( @@ -504,6 +658,7 @@ mod tests { body: Bytes::from_static(b""), metadata: Default::default(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -530,6 +685,7 @@ mod tests { body: Bytes::from_static(b""), metadata: Default::default(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -556,6 +712,7 @@ mod tests { body: Bytes::from_static(b""), metadata: Default::default(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -608,6 +765,7 @@ mod tests { body: Bytes::from_static(b"data"), metadata: Default::default(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -632,6 +790,7 @@ mod tests { body: Bytes::new(), metadata: Default::default(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -673,6 +832,7 @@ mod tests { body: Bytes::from_static(b"0123456789"), metadata: Default::default(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -683,6 +843,7 @@ mod tests { body: Bytes::new(), metadata: Default::default(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -724,6 +885,7 @@ mod tests { body: Bytes::new(), metadata: Default::default(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -745,5 +907,157 @@ mod tests { assert!(c.upload_part_copy); assert!(c.batch_delete); assert!(!c.metadata_in_listings); // mirror standard S3 + assert!(c.object_lock); + } + + // ---- Object Lock ------------------------------------------------------- + // + // These matter because the whole rollback story rests on a committed root + // record being undeletable. If the fake let a retained object be rewritten, + // every rollback test built on it would pass vacuously. + // + // It is *removal* that needs care in the other direction: retention makes + // an object indestructible, not unhideable, and the fake used to claim + // otherwise. See the module documentation. + + fn locked(key: &str, body: &'static [u8], secs: u64) -> PutBlobInput { + PutBlobInput::new(key, Bytes::from_static(body)).with_object_lock(ObjectLock { + mode: super::super::ObjectLockMode::Compliance, + retain_until: SystemTime::now() + std::time::Duration::from_secs(secs), + }) + } + + /// The correction. A `DeleteObject` on a retained object **succeeds** — it + /// writes a delete marker — and every ordinary read then answers + /// "not found" while the bytes sit underneath, undeletable. + /// + /// This test used to assert the delete was refused, which is what a + /// non-versioned bucket would do and what no bucket with Object Lock can + /// be. Verified against MinIO before changing it. + #[tokio::test] + async fn a_retained_object_can_be_hidden_but_not_destroyed() { + let b = MemoryBackend::new(); + b.put_blob(locked("roots/0001", b"root", 3600)) + .await + .unwrap(); + + b.delete_blob("roots/0001").await.expect("S3 allows this"); + + // Hidden from everything that reads the current version. + assert!(matches!( + b.get_blob("roots/0001", None).await, + Err(FsError::NotFound) + )); + assert!(matches!( + b.head_blob("roots/0001").await, + Err(FsError::NotFound) + )); + let listed = b + .list_blobs(ListBlobsInput { + prefix: "roots/", + ..Default::default() + }) + .await + .unwrap(); + assert!(listed.items.is_empty(), "{listed:?}"); + + // And still there, in the only place that matters. + assert_eq!( + b.get_retained_blob("roots/0001").await.unwrap().body, + Bytes::from_static(b"root") + ); + } + + /// The consequence, stated on its own because it is the one that bites: a + /// hidden key accepts a conditional create. `If-None-Match: *` tests the + /// current version, and a delete marker is a current version that is not an + /// object — so a writer using the conditional PUT as its "am I the first?" + /// test is told yes. + #[tokio::test] + async fn a_hidden_key_accepts_a_conditional_create() { + let b = MemoryBackend::new(); + b.put_blob_if_not_exists(locked("origin/receipt", b"first", 3600)) + .await + .unwrap(); + assert!(matches!( + b.put_blob_if_not_exists(locked("origin/receipt", b"second", 3600)) + .await, + Err(FsError::AlreadyExists) + )); + + b.delete_blob("origin/receipt").await.unwrap(); + + b.put_blob_if_not_exists(locked("origin/receipt", b"second", 3600)) + .await + .expect("a delete marker makes the key look untaken"); + } + + #[tokio::test] + async fn retained_object_cannot_be_overwritten() { + let b = MemoryBackend::new(); + b.put_blob(locked("roots/0001", b"root", 3600)) + .await + .unwrap(); + + assert!(matches!( + b.put_blob(PutBlobInput::new( + "roots/0001", + Bytes::from_static(b"forged") + )) + .await, + Err(FsError::AccessDenied) + )); + assert!(matches!( + b.copy_blob(CopyBlobInput { + source_key: "roots/0001".into(), + destination_key: "roots/0001".into(), + replace_metadata: None, + replace_content_type: None, + }) + .await, + Err(FsError::AccessDenied) + )); + assert_eq!( + b.get_blob("roots/0001", None).await.unwrap().body, + Bytes::from_static(b"root") + ); + } + + #[tokio::test] + async fn expired_retention_stops_binding() { + let b = MemoryBackend::new(); + // Retention already in the past — S3 would allow the delete. + b.put_blob( + PutBlobInput::new("roots/0001", Bytes::from_static(b"root")).with_object_lock( + ObjectLock { + mode: super::super::ObjectLockMode::Compliance, + retain_until: SystemTime::now() - std::time::Duration::from_secs(1), + }, + ), + ) + .await + .unwrap(); + + b.delete_blob("roots/0001").await.unwrap(); + assert!(matches!( + b.head_blob("roots/0001").await, + Err(FsError::NotFound) + )); + } + + #[tokio::test] + async fn unlocked_objects_are_unaffected() { + let b = MemoryBackend::new(); + b.put_blob(PutBlobInput::new("slabs/0001", Bytes::from_static(b"data"))) + .await + .unwrap(); + b.put_blob(PutBlobInput::new( + "slabs/0001", + Bytes::from_static(b"data2"), + )) + .await + .unwrap(); + b.delete_blob("slabs/0001").await.unwrap(); + assert_eq!(b.object_count(), 0); } } diff --git a/crates/s3fs-core/src/backend/mod.rs b/crates/s3fs-core/src/backend/mod.rs index eea3c46..ab95f43 100644 --- a/crates/s3fs-core/src/backend/mod.rs +++ b/crates/s3fs-core/src/backend/mod.rs @@ -58,6 +58,48 @@ pub struct Capabilities { /// Standard S3: `false`. Yandex S3: `true`. When false, symlink detection /// via metadata flag costs an extra HEAD on cache miss. pub metadata_in_listings: bool, + /// Backend honours per-object Object Lock retention headers on PUT. + /// Real S3 and MinIO do (on a bucket created with Object Lock enabled); + /// the in-memory fake emulates it. When false, [`PutBlobInput::object_lock`] + /// is rejected rather than silently ignored — an unenforced retention is + /// worse than no retention, because it looks like a guarantee. + pub object_lock: bool, +} + +/// Object Lock retention mode. +/// +/// The distinction matters: `Governance` can be bypassed by a principal +/// holding `s3:BypassGovernanceRetention`, so it protects against accident, +/// not against an adversary. Only `Compliance` is undeletable by every +/// principal including the account root, which is what makes it usable as a +/// rollback anchor. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ObjectLockMode { + Governance, + Compliance, +} + +impl ObjectLockMode { + /// Wire value for `x-amz-object-lock-mode`. + pub fn as_str(self) -> &'static str { + match self { + ObjectLockMode::Governance => "GOVERNANCE", + ObjectLockMode::Compliance => "COMPLIANCE", + } + } +} + +/// Per-object retention to apply at PUT time. +/// +/// Bucket-level default retention achieves the same thing and is simpler to +/// operate; this exists so a caller can lock root records without relying on +/// bucket configuration it may not control. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct ObjectLock { + pub mode: ObjectLockMode, + /// Absolute instant until which the object version cannot be deleted or + /// overwritten. + pub retain_until: SystemTime, } /// Result of a `HeadObject` (or the equivalent metadata view from a list / @@ -121,6 +163,28 @@ pub struct PutBlobInput { pub body: Bytes, pub metadata: HashMap, pub content_type: Option, + /// Retention to stamp on this object version. `None` leaves it to the + /// bucket's default retention configuration (if any). + pub object_lock: Option, +} + +impl PutBlobInput { + /// Plain PUT with no metadata, content type, or retention. + pub fn new(key: impl Into, body: Bytes) -> Self { + Self { + key: key.into(), + body, + metadata: HashMap::new(), + content_type: None, + object_lock: None, + } + } + + /// Stamp per-object retention on this PUT. + pub fn with_object_lock(mut self, lock: ObjectLock) -> Self { + self.object_lock = Some(lock); + self + } } /// Inputs to `copy_blob`. @@ -162,6 +226,28 @@ pub trait Backend: Send + Sync + std::fmt::Debug + 'static { async fn get_blob(&self, key: &str, range: Option>) -> FsResult; + /// Read the **retained** version of an object, seeing past a delete marker. + /// + /// Object Lock protects a *version*. It does not stop a `DeleteObject` + /// without a version id, which inserts a delete marker: `GetObject` then + /// answers `NoSuchKey` while the retained version sits underneath, + /// genuinely undeletable. Confirmed against MinIO — the delete returns + /// `{"DeleteMarker": true}` and succeeds, and deleting the version itself + /// fails with *"Object is WORM protected"*. + /// + /// That gap matters wherever **absence carries meaning**. A substituted + /// object is caught by a signature; a hidden one has no signature to check, + /// so "nobody ever wrote this" and "somebody hid it" become the same + /// answer — and for the boot machine the first authorises creating a new + /// filesystem. This method is how they stay distinguishable. + /// + /// Returns the **oldest** non-marker version: the one the conditional PUT + /// created and retention pinned. Anything newer was written by someone not + /// using a conditional PUT, which is to say not by us. + /// + /// Needs `s3:ListBucketVersions` in addition to `s3:GetObject`. + async fn get_retained_blob(&self, key: &str) -> FsResult; + async fn put_blob(&self, input: PutBlobInput) -> FsResult; /// Atomic create. Returns [`crate::errors::FsError::AlreadyExists`] if the diff --git a/crates/s3fs-core/src/buffer/mod.rs b/crates/s3fs-core/src/buffer/mod.rs deleted file mode 100644 index f19b4ad..0000000 --- a/crates/s3fs-core/src/buffer/mod.rs +++ /dev/null @@ -1,10 +0,0 @@ -//! Buffer-pool layer: per-part dirty bookkeeping (`RangeSet`), part state -//! machine (`PartBuf`), and the memory-bounded `BufferPool`. - -pub mod part; -pub mod pool; -pub mod ranges; - -pub use part::{PartBuf, PartState}; -pub use pool::{BufferPool, PartKey}; -pub use ranges::RangeSet; diff --git a/crates/s3fs-core/src/buffer/part.rs b/crates/s3fs-core/src/buffer/part.rs deleted file mode 100644 index 3a52ee6..0000000 --- a/crates/s3fs-core/src/buffer/part.rs +++ /dev/null @@ -1,400 +0,0 @@ -//! `PartBuf` — per-part buffer with state machine and dirty-range bookkeeping. -//! -//! Each MPU part lives as one `PartBuf`. Writes are absorbed into the buffer -//! and recorded as dirty ranges; the flusher consults `dirty` to decide -//! whether to upload the whole part or read-modify-write a partial part. -//! -//! ```text -//! new_clean ──────────────► Clean ◄──────────────────┐ -//! │ │ -//! │ apply_write │ mark_clean -//! ▼ │ -//! new_empty_dirty ─────────► Dirty ──flush_start──► Flushing -//! ▲ │ -//! │ flush_failed │ flush_ok -//! └─────────────────────┤ -//! ▼ -//! Flushed -//! ``` - -use std::ops::Range; - -use bytes::{Bytes, BytesMut}; - -use super::ranges::RangeSet; -use crate::errors::{FsError, FsResult}; - -/// State machine for a single part. -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum PartState { - /// In sync with what we last fetched from S3 (or, after `mark_clean`, - /// with what we successfully uploaded). - Clean, - /// Local writes have modified bytes in `body`; the dirty subranges live - /// in `PartBuf::dirty`. - Dirty, - /// Currently being uploaded; bytes must not be mutated until the result - /// arrives. - Flushing, - /// `UploadPart` (or `UploadPartCopy`) succeeded; the ETag in - /// `PartBuf::etag` is what `CompleteMultipartUpload` will reference. - Flushed, -} - -/// One part of a file: its byte buffer, dirty bookkeeping, and state. -#[derive(Debug)] -pub struct PartBuf { - /// 0-based part index. S3 wire `PartNumber` is `part_index + 1`. - pub part_index: u32, - /// Tier capacity for this part. The buffer's logical size is bounded by - /// this; the *valid* length may be less (last part of a small file). - pub part_size: u64, - /// File offset of byte 0 of this part. - pub part_start: u64, - /// Number of valid bytes (≤ `part_size`). - pub valid_len: u64, - /// Backing storage. Always exactly `valid_len` bytes long; capacity may - /// be `part_size`-aligned for cheap appends. - body: BytesMut, - /// Current state. - pub state: PartState, - /// Dirty byte ranges *within the part* (offsets are part-relative). - pub dirty: RangeSet, - /// ETag from the last successful upload of this part. `None` until the - /// first successful flush. - pub etag: Option, - /// In-flight background `UploadPart` job, if any. Set by `pwrite` when - /// it makes the part fully dirty and submits it eagerly to the - /// flusher; awaited by the next `sync()` for this file. The `Mutex` - /// is `tokio` (not `parking_lot`) because we need to await it. - pub upload_in_flight: - std::sync::Mutex>>, -} - -impl PartBuf { - /// Construct a `Clean` part from bytes fetched from S3. - /// - /// `body.len()` becomes `valid_len`. `part_size` may be larger (we just - /// remember the tier capacity so future writes at higher offsets within - /// the part can extend.) - pub fn new_clean(part_index: u32, part_size: u64, part_start: u64, body: Bytes) -> Self { - let valid_len = body.len() as u64; - let mut buf = BytesMut::with_capacity(part_size as usize); - buf.extend_from_slice(&body); - Self { - part_index, - part_size, - part_start, - valid_len, - body: buf, - state: PartState::Clean, - dirty: RangeSet::new(), - etag: None, - upload_in_flight: std::sync::Mutex::new(None), - } - } - - /// Construct an empty `Dirty` part. Used when the file is being grown - /// past EOF and the part is not yet present in S3 (no need to fetch - /// before writing). All bytes are initially zero-valued in storage; only - /// bytes covered by `dirty` should be considered authoritative. - pub fn new_empty_dirty(part_index: u32, part_size: u64, part_start: u64) -> Self { - Self { - part_index, - part_size, - part_start, - valid_len: 0, - body: BytesMut::with_capacity(part_size as usize), - state: PartState::Dirty, - dirty: RangeSet::new(), - etag: None, - upload_in_flight: std::sync::Mutex::new(None), - } - } - - /// Take any in-flight upload receiver out of the part. Used by `sync` - /// to await the result of an eager `pwrite`-triggered upload. - pub fn take_inflight( - &self, - ) -> Option> { - self.upload_in_flight.lock().ok().and_then(|mut g| g.take()) - } - - /// Park an in-flight upload receiver on the part. - pub fn park_inflight(&self, rx: tokio::sync::oneshot::Receiver) { - if let Ok(mut g) = self.upload_in_flight.lock() { - *g = Some(rx); - } - } - - /// Read access to the underlying bytes (`valid_len` bytes long). - pub fn body(&self) -> &[u8] { - &self.body - } - - /// Cost in bytes for buffer-pool memory accounting. Counts the allocated - /// capacity, not just `valid_len`, so eviction reflects real RAM use. - pub fn alloc_bytes(&self) -> u64 { - self.body.capacity() as u64 - } - - /// Apply a write at part-relative `offset_in_part`, copying from `data`. - /// - /// Extends `valid_len` if the write reaches past current EOF. Marks the - /// affected range dirty and transitions state from `Clean → Dirty`. - /// Errors if the write would exceed `part_size` or if state is not - /// `Clean`/`Dirty` (i.e. don't write to a part that's `Flushing`). - pub fn apply_write(&mut self, offset_in_part: u64, data: &[u8]) -> FsResult<()> { - match self.state { - PartState::Clean | PartState::Dirty => {} - PartState::Flushing => { - return Err(FsError::WouldBlock); - } - PartState::Flushed => { - // Re-dirtying after flush is allowed — the upload was ours, - // but the user wants to mutate again before commit. - } - } - let end = offset_in_part - .checked_add(data.len() as u64) - .ok_or(FsError::Invalid("write overflow"))?; - if end > self.part_size { - return Err(FsError::Invalid("write past part_size")); - } - // Grow the body if needed. - if end > self.valid_len { - let extra = (end - self.valid_len) as usize; - self.body.extend_from_slice(&vec![0u8; extra]); - self.valid_len = end; - } - let dst = &mut self.body[offset_in_part as usize..end as usize]; - dst.copy_from_slice(data); - self.dirty.insert(offset_in_part..end); - self.state = PartState::Dirty; - self.etag = None; // any prior upload is now stale - Ok(()) - } - - /// Read part-relative bytes `[offset, offset+len)` into a new `Bytes`. - /// Returns up to `valid_len` bytes; if the requested range extends past - /// the valid end, the result is truncated. - pub fn read(&self, offset_in_part: u64, len: u64) -> Bytes { - if offset_in_part >= self.valid_len { - return Bytes::new(); - } - let end = (offset_in_part + len).min(self.valid_len) as usize; - let start = offset_in_part as usize; - Bytes::copy_from_slice(&self.body[start..end]) - } - - /// True iff every byte in `[0, valid_len)` is dirty. Cheap. - pub fn is_fully_dirty(&self) -> bool { - self.dirty.covers_range(0..self.valid_len) - } - - /// Returns the dirty sub-ranges (part-relative). - pub fn dirty_ranges(&self) -> &[Range] { - self.dirty.ranges() - } - - /// Returns the *clean* (non-dirty) sub-ranges within `[0, valid_len)`, - /// i.e. the regions that need to be loaded from S3 during a partial - /// flush's read-modify-write. - pub fn clean_subranges(&self) -> Vec> { - self.dirty.complement_within(0..self.valid_len) - } - - /// Transition: `Dirty` → `Flushing`. The caller (flusher) must have - /// just snapshotted `body` for upload; after this, `apply_write` will - /// fail with `WouldBlock` until the flush completes. - pub fn mark_flushing(&mut self) -> FsResult<()> { - if !matches!(self.state, PartState::Dirty) { - return Err(FsError::Invalid("mark_flushing requires Dirty state")); - } - self.state = PartState::Flushing; - Ok(()) - } - - /// Transition: `Flushing` → `Flushed`. Records the ETag returned by S3. - /// Dirty ranges are cleared (the upload absorbed them). - pub fn mark_flushed(&mut self, etag: String) -> FsResult<()> { - if !matches!(self.state, PartState::Flushing) { - return Err(FsError::Invalid("mark_flushed requires Flushing state")); - } - self.state = PartState::Flushed; - self.etag = Some(etag); - self.dirty.clear(); - Ok(()) - } - - /// Transition: `Flushing` → `Dirty` after a failed upload. Dirty ranges - /// are preserved so the flusher can retry. - pub fn mark_flush_failed(&mut self) -> FsResult<()> { - if !matches!(self.state, PartState::Flushing) { - return Err(FsError::Invalid( - "mark_flush_failed requires Flushing state", - )); - } - self.state = PartState::Dirty; - Ok(()) - } - - /// Transition: `Flushed` → `Clean` after a successful - /// `CompleteMultipartUpload` finalised the part. - pub fn mark_clean(&mut self) -> FsResult<()> { - if !matches!(self.state, PartState::Flushed) { - return Err(FsError::Invalid("mark_clean requires Flushed state")); - } - self.state = PartState::Clean; - // ETag is preserved — it's what `commit` will pass back if we ever - // re-fetch and want optimistic concurrency. - Ok(()) - } - - /// Snapshot the body as immutable `Bytes` for upload. Cheap (`Bytes::copy_from_slice`). - pub fn snapshot(&self) -> Bytes { - Bytes::copy_from_slice(&self.body) - } -} - -#[cfg(test)] -#[allow(clippy::single_range_in_vec_init)] // `&[a..b]` is a one-element slice of Range — intentional in these asserts -mod tests { - use super::*; - - fn fresh_clean() -> PartBuf { - PartBuf::new_clean(0, 16, 0, Bytes::from_static(b"0123456789abcdef")) - } - - #[test] - fn new_clean_snapshot_matches_input() { - let p = fresh_clean(); - assert_eq!(p.state, PartState::Clean); - assert_eq!(p.valid_len, 16); - assert_eq!(p.body(), b"0123456789abcdef"); - assert!(p.dirty.is_empty()); - assert!(!p.is_fully_dirty()); - } - - #[test] - fn new_empty_dirty_starts_zero_len() { - let p = PartBuf::new_empty_dirty(0, 16, 0); - assert_eq!(p.state, PartState::Dirty); - assert_eq!(p.valid_len, 0); - assert!(p.body().is_empty()); - } - - #[test] - fn apply_write_marks_dirty_and_grows_body() { - let mut p = PartBuf::new_empty_dirty(0, 16, 0); - p.apply_write(0, b"abcd").unwrap(); - assert_eq!(p.state, PartState::Dirty); - assert_eq!(p.valid_len, 4); - assert_eq!(p.body(), b"abcd"); - assert_eq!(p.dirty_ranges(), &[0..4]); - } - - #[test] - fn apply_write_within_clean_part_marks_dirty() { - let mut p = fresh_clean(); - p.apply_write(4, b"XYZ").unwrap(); - assert_eq!(p.state, PartState::Dirty); - assert_eq!(p.body(), b"0123XYZ789abcdef"); - assert_eq!(p.dirty_ranges(), &[4..7]); - } - - #[test] - fn apply_write_past_part_size_errors() { - let mut p = fresh_clean(); - let err = p.apply_write(15, b"XX").unwrap_err(); - assert!(matches!(err, FsError::Invalid(_))); - } - - #[test] - fn apply_write_while_flushing_returns_would_block() { - let mut p = PartBuf::new_empty_dirty(0, 16, 0); - p.apply_write(0, b"abcd").unwrap(); - p.mark_flushing().unwrap(); - let err = p.apply_write(4, b"x").unwrap_err(); - assert!(matches!(err, FsError::WouldBlock)); - } - - #[test] - fn read_returns_valid_subrange() { - let p = fresh_clean(); - assert_eq!(p.read(2, 4)[..], b"2345"[..]); - // Past valid_len gets clipped. - assert_eq!(p.read(14, 10)[..], b"ef"[..]); - // Wholly past valid_len returns empty. - assert!(p.read(20, 5).is_empty()); - } - - #[test] - fn fully_dirty_after_full_overwrite() { - let mut p = PartBuf::new_clean(0, 16, 0, Bytes::from_static(b"0123456789abcdef")); - p.apply_write(0, b"FFFFFFFFFFFFFFFF").unwrap(); - assert!(p.is_fully_dirty()); - assert_eq!(p.dirty_ranges(), &[0..16]); - } - - #[test] - fn clean_subranges_are_complement_of_dirty() { - let mut p = PartBuf::new_clean(0, 16, 0, Bytes::from_static(b"0123456789abcdef")); - p.apply_write(2, b"X").unwrap(); - p.apply_write(8, b"Y").unwrap(); - // dirty: 2..3, 8..9 → clean: 0..2, 3..8, 9..16 - assert_eq!(p.clean_subranges(), vec![0..2, 3..8, 9..16]); - } - - #[test] - fn state_transitions_happy_path() { - let mut p = PartBuf::new_empty_dirty(0, 16, 0); - p.apply_write(0, b"abcd").unwrap(); - p.mark_flushing().unwrap(); - assert_eq!(p.state, PartState::Flushing); - p.mark_flushed("etag-v1".into()).unwrap(); - assert_eq!(p.state, PartState::Flushed); - assert_eq!(p.etag.as_deref(), Some("etag-v1")); - assert!(p.dirty.is_empty()); - p.mark_clean().unwrap(); - assert_eq!(p.state, PartState::Clean); - } - - #[test] - fn state_transitions_failure_path() { - let mut p = PartBuf::new_empty_dirty(0, 16, 0); - p.apply_write(0, b"abcd").unwrap(); - p.mark_flushing().unwrap(); - p.mark_flush_failed().unwrap(); - // Returned to Dirty, dirty ranges preserved for retry. - assert_eq!(p.state, PartState::Dirty); - assert_eq!(p.dirty_ranges(), &[0..4]); - } - - #[test] - fn redirty_after_flushed_clears_etag() { - let mut p = PartBuf::new_empty_dirty(0, 16, 0); - p.apply_write(0, b"abcd").unwrap(); - p.mark_flushing().unwrap(); - p.mark_flushed("etag-v1".into()).unwrap(); - // Now apply another write — should re-dirty, drop etag. - p.apply_write(4, b"ef").unwrap(); - assert_eq!(p.state, PartState::Dirty); - assert!(p.etag.is_none()); - } - - #[test] - fn invalid_state_transitions_error() { - let mut p = PartBuf::new_empty_dirty(0, 16, 0); - // Can't mark_flushing from Dirty-but-empty? Actually we can — Dirty - // state allows flushing; let's test something that's truly invalid. - assert!(p.mark_flushed("nope".into()).is_err()); // Dirty → not allowed - assert!(p.mark_clean().is_err()); // Dirty → not allowed - } - - #[test] - fn alloc_bytes_reflects_capacity() { - let p = PartBuf::new_empty_dirty(0, 16, 0); - assert!(p.alloc_bytes() >= 16); - } -} diff --git a/crates/s3fs-core/src/buffer/pool.rs b/crates/s3fs-core/src/buffer/pool.rs deleted file mode 100644 index 24861e1..0000000 --- a/crates/s3fs-core/src/buffer/pool.rs +++ /dev/null @@ -1,329 +0,0 @@ -//! `BufferPool` — memory-bounded cache of `PartBuf`s, keyed by `(inode, part)`. -//! -//! The pool's only job is in-memory accounting and LRU eviction of *clean* -//! parts. Dirty / in-flight parts are pinned (cannot be evicted). Fetching -//! parts from S3 is the caller's responsibility — this module never touches -//! the backend. -//! -//! Eviction policy: -//! - Targets (`memory_limit_bytes`) come from `Config`. -//! - Eviction runs at insert-time when adding the new part would exceed the -//! limit. Walks LRU oldest-first, dropping `Clean` and `Flushed` parts -//! until headroom is reclaimed. If we can't reclaim enough (everything -//! pinned), the insert succeeds anyway and `used > limit` becomes true; -//! future flushes are responsible for relieving pressure. - -use std::sync::Arc; - -use hashlink::LinkedHashMap; -use parking_lot::{Mutex, RwLock}; - -use super::part::{PartBuf, PartState}; -use crate::config::Config; -use crate::inode::InodeId; - -/// Identifies one part in the pool. -#[derive(Debug, Clone, Copy, Hash, PartialEq, Eq)] -pub struct PartKey { - pub inode_id: InodeId, - pub part_index: u32, -} - -impl PartKey { - pub fn new(inode_id: InodeId, part_index: u32) -> Self { - Self { - inode_id, - part_index, - } - } -} - -#[derive(Debug)] -struct Inner { - used: u64, - /// Insertion order = age. Move-to-back on access. - lru: LinkedHashMap>>, -} - -/// In-memory part cache. -#[derive(Debug)] -pub struct BufferPool { - config: Arc, - inner: Mutex, -} - -impl BufferPool { - pub fn new(config: Arc) -> Arc { - Arc::new(Self { - config, - inner: Mutex::new(Inner { - used: 0, - lru: LinkedHashMap::new(), - }), - }) - } - - pub fn limit(&self) -> u64 { - self.config.memory_limit_bytes - } - - pub fn used(&self) -> u64 { - self.inner.lock().used - } - - pub fn len(&self) -> usize { - self.inner.lock().lru.len() - } - - pub fn is_empty(&self) -> bool { - self.len() == 0 - } - - /// Look up a part. On hit, marks it most-recently-used. - pub fn get(&self, key: PartKey) -> Option>> { - let mut g = self.inner.lock(); - // hashlink's `to_back` returns the value reference; use raw_entry-ish dance - // via remove + reinsert (cheap; the value is an Arc clone). - let v = g.lru.remove(&key)?; - g.lru.insert(key, v.clone()); - Some(v) - } - - /// Cheap probe: does the pool currently hold this key? Doesn't bump LRU. - pub fn contains(&self, key: PartKey) -> bool { - self.inner.lock().lru.contains_key(&key) - } - - /// Install a fresh `PartBuf` for `key`. Returns the wrapped `Arc`. - /// If a part already exists for `key`, the existing entry is replaced - /// and its memory is subtracted before the new entry's memory is added. - /// - /// Before inserting, evicts clean parts in LRU order to make headroom. - pub fn insert(&self, key: PartKey, part: PartBuf) -> Arc> { - let new_cost = part.alloc_bytes(); - let mut g = self.inner.lock(); - - if let Some(existing) = g.lru.remove(&key) { - g.used = g.used.saturating_sub(existing.read().alloc_bytes()); - } - - // Evict clean parts until we'd fit, or we run out of evictable parts. - if g.used + new_cost > self.config.memory_limit_bytes { - let target = self.config.memory_limit_bytes.saturating_sub(new_cost); - self.evict_clean_locked(&mut g, target); - } - - g.used = g.used.saturating_add(new_cost); - let arc = Arc::new(RwLock::new(part)); - g.lru.insert(key, arc.clone()); - arc - } - - /// Drop a part unconditionally — used on file unlink / abort. Returns - /// `true` if anything was removed. - pub fn forget(&self, key: PartKey) -> bool { - let mut g = self.inner.lock(); - if let Some(v) = g.lru.remove(&key) { - g.used = g.used.saturating_sub(v.read().alloc_bytes()); - true - } else { - false - } - } - - /// Forget every part owned by a given inode. Returns the count removed. - pub fn forget_inode(&self, inode_id: InodeId) -> usize { - let mut g = self.inner.lock(); - let to_drop: Vec = g - .lru - .iter() - .filter(|(k, _)| k.inode_id == inode_id) - .map(|(k, _)| *k) - .collect(); - let n = to_drop.len(); - for k in to_drop { - if let Some(v) = g.lru.remove(&k) { - g.used = g.used.saturating_sub(v.read().alloc_bytes()); - } - } - n - } - - /// Evict clean / flushed parts in LRU order until `used` drops to - /// `target` or fewer bytes. Pinned parts (`Dirty`, `Flushing`) are - /// skipped. Returns the number evicted. - pub fn evict_clean(&self, target: u64) -> usize { - let mut g = self.inner.lock(); - self.evict_clean_locked(&mut g, target) - } - - fn evict_clean_locked(&self, g: &mut Inner, target: u64) -> usize { - let mut n = 0; - // Walk LRU oldest-first. - let candidates: Vec = g.lru.iter().map(|(k, _)| *k).collect(); - for k in candidates { - if g.used <= target { - break; - } - let evictable = match g.lru.get(&k) { - None => continue, - Some(part) => { - let st = part.read().state.clone(); - matches!(st, PartState::Clean | PartState::Flushed) - } - }; - if !evictable { - continue; - } - if let Some(part) = g.lru.remove(&k) { - g.used = g.used.saturating_sub(part.read().alloc_bytes()); - n += 1; - } - } - n - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::inode::InodeId; - use bytes::Bytes; - use std::num::NonZeroU64; - - fn key(ino: u64, part: u32) -> PartKey { - PartKey::new(InodeId(NonZeroU64::new(ino).unwrap()), part) - } - - fn small_pool(limit: u64) -> Arc { - let cfg = Arc::new(Config::builder().memory_limit_bytes(limit).build()); - BufferPool::new(cfg) - } - - fn clean_part(part_index: u32, body: &[u8], part_size: u64) -> PartBuf { - PartBuf::new_clean(part_index, part_size, 0, Bytes::copy_from_slice(body)) - } - - #[test] - fn insert_and_get_basic() { - let p = small_pool(1024); - let k = key(1, 0); - p.insert(k, clean_part(0, &[1u8; 64], 64)); - assert_eq!(p.len(), 1); - let part = p.get(k).expect("hit"); - assert_eq!(part.read().valid_len, 64); - } - - #[test] - fn get_misses_when_absent() { - let p = small_pool(1024); - assert!(p.get(key(1, 0)).is_none()); - } - - #[test] - fn insert_overwrites_and_recomputes_memory() { - let p = small_pool(1024); - let k = key(1, 0); - p.insert(k, clean_part(0, &[0u8; 64], 64)); - let used_after_first = p.used(); - p.insert(k, clean_part(0, &[0u8; 128], 128)); - assert!(p.used() > used_after_first); - assert_eq!(p.len(), 1); - } - - #[test] - fn forget_removes_and_credits_memory() { - let p = small_pool(1024); - let k = key(1, 0); - p.insert(k, clean_part(0, &[0u8; 64], 64)); - assert!(p.forget(k)); - assert!(p.is_empty()); - assert_eq!(p.used(), 0); - // Idempotent. - assert!(!p.forget(k)); - } - - #[test] - fn forget_inode_drops_all_parts_of_one_inode() { - let p = small_pool(4096); - for i in 0..5 { - p.insert(key(1, i), clean_part(i, &[0u8; 64], 64)); - } - for i in 0..5 { - p.insert(key(2, i), clean_part(i, &[0u8; 64], 64)); - } - assert_eq!(p.len(), 10); - let dropped = p.forget_inode(InodeId(NonZeroU64::new(1).unwrap())); - assert_eq!(dropped, 5); - assert_eq!(p.len(), 5); - } - - #[test] - fn lru_promotes_on_get() { - // Limit fits two 64-byte parts comfortably but a third triggers - // eviction. After touching k0, k1 becomes the LRU victim. - let p = small_pool(150); - let k0 = key(1, 0); - let k1 = key(1, 1); - let k2 = key(1, 2); - p.insert(k0, clean_part(0, &[0u8; 64], 64)); - p.insert(k1, clean_part(1, &[0u8; 64], 64)); - let _ = p.get(k0); // promote k0 - p.insert(k2, clean_part(2, &[0u8; 64], 64)); - assert!(p.contains(k0)); - assert!(!p.contains(k1)); - assert!(p.contains(k2)); - } - - #[test] - fn dirty_parts_are_not_evicted() { - // Insert a Dirty part, then try to evict. - let p = small_pool(64); - let k = key(1, 0); - let mut dirty = PartBuf::new_empty_dirty(0, 64, 0); - dirty.apply_write(0, &[7u8; 32]).unwrap(); - p.insert(k, dirty); - // Force eviction with a small target. - let evicted = p.evict_clean(0); - assert_eq!(evicted, 0); - assert!(p.contains(k)); - // used > limit is allowed when only dirty parts exist. - } - - #[test] - fn flushed_parts_are_evictable() { - let p = small_pool(64); - let k = key(1, 0); - let mut part = PartBuf::new_empty_dirty(0, 64, 0); - part.apply_write(0, &[1u8; 32]).unwrap(); - part.mark_flushing().unwrap(); - part.mark_flushed("etag-1".into()).unwrap(); - p.insert(k, part); - let evicted = p.evict_clean(0); - assert_eq!(evicted, 1); - assert!(p.is_empty()); - } - - #[test] - fn insert_pressure_evicts_oldest_clean_first() { - // Two clean parts + one dirty; insert a new clean → dirty must remain. - let p = small_pool(192); - let k0 = key(1, 0); - let k1 = key(1, 1); - let k_dirty = key(1, 2); - p.insert(k0, clean_part(0, &[0u8; 64], 64)); - let mut d = PartBuf::new_empty_dirty(2, 64, 0); - d.apply_write(0, &[5u8; 32]).unwrap(); - p.insert(k_dirty, d); - p.insert(k1, clean_part(1, &[0u8; 64], 64)); - - // Now insert a 4th. The dirty one must survive; one of k0/k1 evicted. - let k_new = key(1, 3); - p.insert(k_new, clean_part(3, &[0u8; 64], 64)); - assert!(p.contains(k_dirty)); - assert!(p.contains(k_new)); - // At least one of k0/k1 was evicted to make room. - let still = p.contains(k0) as u8 + p.contains(k1) as u8; - assert!(still <= 1); - } -} diff --git a/crates/s3fs-core/src/buffer/ranges.rs b/crates/s3fs-core/src/buffer/ranges.rs deleted file mode 100644 index fd9b64c..0000000 --- a/crates/s3fs-core/src/buffer/ranges.rs +++ /dev/null @@ -1,275 +0,0 @@ -//! `RangeSet` — a small set of disjoint, sorted, half-open `u64` ranges. -//! -//! Used to track which byte ranges within a part are dirty (so that a partial -//! flush knows what to read-modify-write) and to compute the *unmodified* -//! complement that needs `UploadPartCopy` during `commit`. -//! -//! Performance is `O(n)` per insert in the worst case, where `n` is the -//! current number of disjoint ranges. For real workloads `n` is small -//! (sequential writers produce 1; even random writers cluster). - -use std::ops::Range; - -/// Set of disjoint `u64` byte ranges, kept sorted by `start` and merged on -/// adjacency (half-open: `r1.end == r2.start` merges). -#[derive(Debug, Default, Clone, PartialEq, Eq)] -pub struct RangeSet { - ranges: Vec>, -} - -impl RangeSet { - pub fn new() -> Self { - Self::default() - } - - pub fn is_empty(&self) -> bool { - self.ranges.is_empty() - } - - pub fn len(&self) -> usize { - self.ranges.len() - } - - /// Borrow the underlying disjoint sorted ranges. - pub fn ranges(&self) -> &[Range] { - &self.ranges - } - - /// Total byte coverage across all ranges. - pub fn total_bytes(&self) -> u64 { - self.ranges.iter().map(|r| r.end - r.start).sum() - } - - /// Drop all ranges. - pub fn clear(&mut self) { - self.ranges.clear(); - } - - /// Insert `range`, merging with any overlapping or adjacent existing - /// ranges. Empty ranges (`start == end`) are silently ignored. - pub fn insert(&mut self, range: Range) { - if range.start >= range.end { - return; - } - // Find the insertion span: ranges that overlap or are adjacent to - // `range`. Those get coalesced into one. - let mut new_start = range.start; - let mut new_end = range.end; - // Index range to remove. - let mut first_to_remove: Option = None; - let mut last_to_remove: Option = None; - - for (i, r) in self.ranges.iter().enumerate() { - // r ends strictly before `new_start` (no overlap, no adjacency) - if r.end < new_start { - continue; - } - // r starts strictly after `new_end` (no overlap, no adjacency) - if r.start > new_end { - break; - } - // Overlap or adjacency; absorb r. - new_start = new_start.min(r.start); - new_end = new_end.max(r.end); - first_to_remove.get_or_insert(i); - last_to_remove = Some(i); - } - - let merged = Range { - start: new_start, - end: new_end, - }; - match (first_to_remove, last_to_remove) { - (Some(a), Some(b)) => { - // Replace the run [a..=b] with the merged range. - self.ranges.splice(a..=b, std::iter::once(merged)); - } - _ => { - // No absorption: insert at the right position to keep sorted. - let pos = self.ranges.partition_point(|r| r.end < merged.start); - self.ranges.insert(pos, merged); - } - } - } - - /// Returns `true` iff every byte in `bound` is covered by some range. - pub fn covers_range(&self, bound: Range) -> bool { - if bound.start >= bound.end { - return true; - } - for r in &self.ranges { - if r.start <= bound.start && r.end >= bound.end { - return true; - } - if r.start > bound.start { - return false; - } - } - false - } - - /// Returns the sub-ranges of `bound` that are **not** covered, in order. - /// These are the byte ranges a partial-part flush must fetch from S3 to - /// fill the gaps before re-uploading. - pub fn complement_within(&self, bound: Range) -> Vec> { - let mut gaps = Vec::new(); - if bound.start >= bound.end { - return gaps; - } - let mut cursor = bound.start; - for r in &self.ranges { - if r.end <= cursor { - continue; - } - if r.start >= bound.end { - break; - } - if r.start > cursor { - gaps.push(Range { - start: cursor, - end: r.start.min(bound.end), - }); - } - cursor = cursor.max(r.end); - if cursor >= bound.end { - break; - } - } - if cursor < bound.end { - gaps.push(Range { - start: cursor, - end: bound.end, - }); - } - gaps - } -} - -#[cfg(test)] -#[allow(clippy::single_range_in_vec_init)] // `&[a..b]` is a one-element slice of Range — intentional -mod tests { - use super::*; - - #[test] - fn empty_set() { - let s = RangeSet::new(); - assert!(s.is_empty()); - assert_eq!(s.total_bytes(), 0); - assert!(s.ranges().is_empty()); - } - - #[test] - fn insert_disjoint_keeps_sorted() { - let mut s = RangeSet::new(); - s.insert(20..30); - s.insert(0..10); - s.insert(50..60); - assert_eq!(s.ranges(), &[0..10, 20..30, 50..60]); - assert_eq!(s.total_bytes(), 30); - } - - #[test] - fn insert_merges_overlapping() { - let mut s = RangeSet::new(); - s.insert(0..10); - s.insert(5..15); // overlaps - assert_eq!(s.ranges(), &[0..15]); - } - - #[test] - fn insert_merges_adjacent() { - let mut s = RangeSet::new(); - s.insert(0..10); - s.insert(10..20); // touches at 10 - assert_eq!(s.ranges(), &[0..20]); - } - - #[test] - fn insert_swallows_run_of_ranges() { - let mut s = RangeSet::new(); - s.insert(0..5); - s.insert(10..15); - s.insert(20..25); - s.insert(2..23); // covers parts of all three - assert_eq!(s.ranges(), &[0..25]); - } - - #[test] - fn insert_inside_existing_range_noop() { - let mut s = RangeSet::new(); - s.insert(0..100); - s.insert(40..50); - assert_eq!(s.ranges(), &[0..100]); - } - - #[test] - fn empty_insert_is_ignored() { - let mut s = RangeSet::new(); - s.insert(10..10); - // `start > end` is also defensively ignored, but constructing one - // trips clippy::reversed_empty_ranges; the start==end case above is - // sufficient to exercise the early-return path. - assert!(s.is_empty()); - } - - #[test] - fn covers_range_strict() { - let mut s = RangeSet::new(); - s.insert(0..100); - assert!(s.covers_range(0..100)); - assert!(s.covers_range(10..50)); - assert!(s.covers_range(99..100)); - assert!(!s.covers_range(0..101)); - assert!(!s.covers_range(100..101)); - } - - #[test] - fn complement_within_full_coverage_is_empty() { - let mut s = RangeSet::new(); - s.insert(0..100); - let gaps = s.complement_within(0..100); - assert!(gaps.is_empty()); - } - - #[test] - fn complement_within_partial_coverage() { - let mut s = RangeSet::new(); - s.insert(10..20); - s.insert(40..50); - // bound 0..60 → gaps: 0..10, 20..40, 50..60 - let gaps = s.complement_within(0..60); - assert_eq!(gaps, vec![0..10, 20..40, 50..60]); - } - - #[test] - fn complement_within_clipped_to_bound() { - let mut s = RangeSet::new(); - s.insert(0..100); - // bound entirely inside an existing range → no gaps - assert!(s.complement_within(20..80).is_empty()); - // bound outside → entire bound is a gap - let mut s2 = RangeSet::new(); - s2.insert(0..10); - assert_eq!(s2.complement_within(50..60), vec![50..60]); - } - - #[test] - fn complement_within_with_adjacency() { - let mut s = RangeSet::new(); - s.insert(0..10); - s.insert(20..30); - // bound exactly hits the boundaries of dirty ranges - let gaps = s.complement_within(10..20); - assert_eq!(gaps, vec![10..20]); - } - - #[test] - fn clear_resets() { - let mut s = RangeSet::new(); - s.insert(0..10); - s.insert(20..30); - s.clear(); - assert!(s.is_empty()); - assert_eq!(s.total_bytes(), 0); - } -} diff --git a/crates/s3fs-core/src/config.rs b/crates/s3fs-core/src/config.rs index 5c76dfb..8850217 100644 --- a/crates/s3fs-core/src/config.rs +++ b/crates/s3fs-core/src/config.rs @@ -1,218 +1,101 @@ //! Engine configuration. //! -//! `Config` is the single source of truth for tuning knobs (part sizes, memory -//! caps, parallelism, timeouts). Construct with [`Config::builder`]. +//! The tuning knobs that used to live here — multipart part schedules, buffer +//! pool budgets, upload parallelism — belonged to the old path-to-key engine +//! and are gone with it. What remains is the mount's own shape plus +//! [`StoreConfig`], which owns everything about the block store. -use std::path::PathBuf; -use std::time::Duration; +use std::sync::Arc; -/// Tiered multipart-upload part schedule. -/// -/// S3 caps an MPU at 10 000 parts. To support large files without paying the -/// 50 GiB-per-file price of using 5 MiB parts everywhere, we use a tiered -/// schedule: small parts at the start of the file (so small files / appends -/// stay cheap), larger parts further in. -/// -/// The default schedule allows files up to ~1.03 TiB: -/// - parts 0..1000: 5 MiB each → 5 GiB -/// - parts 1000..2000: 25 MiB each → 25 GiB additional -/// - parts 2000..10000: 125 MiB each → 1000 GiB additional -#[derive(Debug, Clone)] -pub struct PartSchedule { - /// Each tier is `(part_size_bytes, part_count)`. The `part_count` of the - /// final tier is also the cap on total part count (S3's hard limit is - /// 10 000). - pub tiers: Vec<(u64, u32)>, -} - -impl Default for PartSchedule { - fn default() -> Self { - Self { - tiers: vec![ - (5 * 1024 * 1024, 1000), // 5 MiB × 1000 → 5 GiB - (25 * 1024 * 1024, 1000), // 25 MiB × 1000 → 25 GiB - (125 * 1024 * 1024, 8000), // 125 MiB × 8000 → 1000 GiB - ], - } - } -} +use crate::store::StoreConfig; -impl PartSchedule { - /// Maximum file size representable with this schedule. - pub fn max_file_size(&self) -> u64 { - self.tiers.iter().map(|(sz, n)| sz * (*n as u64)).sum() - } - - /// Total part count across all tiers (must be ≤ 10 000 per S3). - pub fn total_parts(&self) -> u32 { - self.tiers.iter().map(|(_, n)| *n).sum() - } - - /// Reverse direction: given a part index, return its byte range - /// `[start, end)` in the file. Returns `None` if `part_index` is past the - /// schedule's last tier. - pub fn part_range(&self, part_index: u32) -> Option> { - let mut accum_parts: u32 = 0; - let mut accum_bytes: u64 = 0; - for &(part_size, count) in &self.tiers { - if part_index < accum_parts + count { - let into_tier = (part_index - accum_parts) as u64; - let start = accum_bytes + into_tier * part_size; - let end = start + part_size; - return Some(start..end); - } - accum_parts += count; - accum_bytes += part_size * (count as u64); - } - None - } +/// Default cap on symlink hops during one path resolution. +pub const DEFAULT_MAX_SYMLINK_DEPTH: u32 = 40; - /// Map a byte offset to its `(part_index, part_size, part_offset_in_file)`. - /// `part_index` is 0-based; S3's `PartNumber` is `part_index + 1`. - /// Returns `None` if the offset exceeds the schedule's representable range. - pub fn locate(&self, offset: u64) -> Option { - let mut accum_parts: u32 = 0; - let mut accum_bytes: u64 = 0; - for &(part_size, count) in &self.tiers { - let tier_bytes = part_size * (count as u64); - if offset < accum_bytes + tier_bytes { - let into_tier = offset - accum_bytes; - let part_in_tier = (into_tier / part_size) as u32; - let part_index = accum_parts + part_in_tier; - let part_start = accum_bytes + (part_in_tier as u64) * part_size; - return Some(PartLocation { - part_index, - part_size, - part_start, - }); - } - accum_parts += count; - accum_bytes += tier_bytes; - } - None - } -} - -/// Location of a byte within the tiered MPU schedule. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub struct PartLocation { - /// Zero-based part index. S3's wire `PartNumber` is `part_index + 1`. - pub part_index: u32, - /// Part size in bytes for the tier this part belongs to. - pub part_size: u64, - /// Offset of this part's first byte within the file. - pub part_start: u64, -} - -/// Engine configuration. Defaults are tuned for a single 8-core host with -/// ~64 MiB available for the buffer pool. +/// Engine configuration. #[derive(Debug, Clone)] pub struct Config { - /// Tiered MPU part schedule. Defaults to ~1.03 TiB max file size. - pub part_schedule: PartSchedule, - /// Files smaller than this go via single `PutObject`, skipping MPU entirely. - /// Defaults to 5 MiB (one S3 part). - pub single_part_threshold: u64, - /// Max in-flight `UploadPart` requests per file. - pub max_parallel_parts: usize, - /// Max in-flight `UploadPartCopy` requests per `commit`. - pub max_parallel_copy: usize, - /// Max bytes a single `UploadPartCopy` will cover when merging unchanged - /// ranges. Larger merges reduce the number of copy parts but each - /// `UploadPartCopy` is capped at 5 GiB by S3. - pub max_merge_copy_bytes: u64, - /// Total buffer-pool memory cap. Bytes pinned by dirty pages count - /// against this; eviction targets clean pages first. - pub memory_limit_bytes: u64, - /// If `Some`, clean pages may spill to this directory when memory pressure - /// exceeds the cap. Disabled by default — enclave use cases prefer no - /// disk. - pub disk_overflow_dir: Option, - /// Sequential-read prefetch distance, in chunks. - pub read_ahead_chunks: usize, - /// Symlink resolution recursion limit. Linux uses 40. - pub max_symlink_depth: u32, - /// Per-S3-request timeout. - pub request_timeout: Duration, - /// How long cached file/directory attrs are considered fresh before the - /// next op triggers a re-validation. - pub attr_cache_ttl: Duration, - /// Path the guest sees for the single preopen. Defaults to `/`. + /// Path the guest sees as its preopen root. pub mount_path: String, - /// Optional key prefix inside the bucket. All operations are relative to - /// `bucket/{prefix}`. Empty = bucket root. - pub bucket_prefix: String, + /// How many symlinks one resolution may follow before giving up. Bounds + /// the work a malicious or merely circular symlink graph can cause. + pub max_symlink_depth: u32, + /// Block store settings. + pub store: StoreConfig, } impl Default for Config { fn default() -> Self { - Self { - part_schedule: PartSchedule::default(), - single_part_threshold: 5 * 1024 * 1024, // 5 MiB - max_parallel_parts: 8, - max_parallel_copy: 8, - max_merge_copy_bytes: 128 * 1024 * 1024, // 128 MiB - memory_limit_bytes: 64 * 1024 * 1024, // 64 MiB - disk_overflow_dir: None, - read_ahead_chunks: 2, - max_symlink_depth: 40, - request_timeout: Duration::from_secs(30), - attr_cache_ttl: Duration::from_secs(1), + Config { mount_path: "/".to_string(), - bucket_prefix: String::new(), + max_symlink_depth: DEFAULT_MAX_SYMLINK_DEPTH, + store: StoreConfig::default(), } } } impl Config { - /// Start from defaults; mutate via the returned builder. pub fn builder() -> ConfigBuilder { ConfigBuilder { - inner: Self::default(), + config: Config::default(), } } + + pub fn record_size(&self) -> usize { + self.store.record_size + } + + pub fn shared(self) -> Arc { + Arc::new(self) + } } -/// Fluent builder over [`Config`]. -#[derive(Debug, Clone)] +/// Fluent builder for [`Config`]. +#[derive(Debug)] pub struct ConfigBuilder { - inner: Config, + config: Config, } -macro_rules! setter { - ($name:ident, $ty:ty) => { - pub fn $name(mut self, v: $ty) -> Self { - self.inner.$name = v; - self +impl ConfigBuilder { + pub fn mount_path(mut self, path: impl Into) -> Self { + self.config.mount_path = path.into(); + self + } + + /// Key prefix inside both buckets. + pub fn bucket_prefix(mut self, prefix: impl Into) -> Self { + let mut prefix = prefix.into(); + // A prefix is a path component, not a substring: without the + // separator, prefix "a" would also match keys under "ab/". + if !prefix.is_empty() && !prefix.ends_with('/') { + prefix.push('/'); } - }; -} + self.config.store.prefix = prefix; + self + } -impl ConfigBuilder { - setter!(part_schedule, PartSchedule); - setter!(single_part_threshold, u64); - setter!(max_parallel_parts, usize); - setter!(max_parallel_copy, usize); - setter!(max_merge_copy_bytes, u64); - setter!(memory_limit_bytes, u64); - setter!(disk_overflow_dir, Option); - setter!(read_ahead_chunks, usize); - setter!(max_symlink_depth, u32); - setter!(request_timeout, Duration); - setter!(attr_cache_ttl, Duration); + pub fn record_size(mut self, bytes: usize) -> Self { + self.config.store.record_size = bytes; + self + } - pub fn mount_path(mut self, v: impl Into) -> Self { - self.inner.mount_path = v.into(); + pub fn max_symlink_depth(mut self, depth: u32) -> Self { + self.config.max_symlink_depth = depth; self } - pub fn bucket_prefix(mut self, v: impl Into) -> Self { - self.inner.bucket_prefix = v.into(); + pub fn block_cache_bytes(mut self, bytes: u64) -> Self { + self.config.store.block_cache_bytes = bytes; + self + } + + pub fn root_retention(mut self, retention: Option) -> Self { + self.config.store.root_retention = retention; self } pub fn build(self) -> Config { - self.inner + self.config } } @@ -221,77 +104,42 @@ mod tests { use super::*; #[test] - fn default_schedule_caps_at_about_one_tib() { - let s = PartSchedule::default(); - // 5 GiB + 25 GiB + 1000 GiB = 1030 GiB ≈ 1.006 TiB - assert_eq!( - s.max_file_size(), - 5u64 * 1024 * 1024 * 1000 + 25u64 * 1024 * 1024 * 1000 + 125u64 * 1024 * 1024 * 8000 - ); - assert_eq!(s.total_parts(), 10_000); - } - - #[test] - fn locate_in_first_tier() { - let s = PartSchedule::default(); - let loc = s.locate(0).unwrap(); - assert_eq!(loc.part_index, 0); - assert_eq!(loc.part_size, 5 * 1024 * 1024); - assert_eq!(loc.part_start, 0); - - let loc = s.locate(5 * 1024 * 1024).unwrap(); - assert_eq!(loc.part_index, 1); - assert_eq!(loc.part_start, 5 * 1024 * 1024); + fn defaults_are_usable() { + let c = Config::default(); + assert_eq!(c.mount_path, "/"); + assert_eq!(c.max_symlink_depth, DEFAULT_MAX_SYMLINK_DEPTH); + c.store.validate().unwrap(); } + /// A prefix names a directory-like scope. Without the separator, prefix + /// "tenant" would also cover keys belonging to "tenant-other". #[test] - fn locate_at_tier_boundaries() { - let s = PartSchedule::default(); - let first_tier_end = 5u64 * 1024 * 1024 * 1000; - let loc = s.locate(first_tier_end).unwrap(); - assert_eq!(loc.part_index, 1000); // first part of tier 2 - assert_eq!(loc.part_size, 25 * 1024 * 1024); - - let second_tier_end = first_tier_end + 25u64 * 1024 * 1024 * 1000; - let loc = s.locate(second_tier_end).unwrap(); - assert_eq!(loc.part_index, 2000); // first part of tier 3 - assert_eq!(loc.part_size, 125 * 1024 * 1024); - } + fn bucket_prefix_gains_a_trailing_separator() { + let c = Config::builder().bucket_prefix("tenant").build(); + assert_eq!(c.store.prefix, "tenant/"); - #[test] - fn locate_beyond_max_returns_none() { - let s = PartSchedule::default(); - assert!(s.locate(s.max_file_size()).is_none()); - assert!(s.locate(u64::MAX).is_none()); - } + let c = Config::builder().bucket_prefix("tenant/").build(); + assert_eq!(c.store.prefix, "tenant/"); - #[test] - fn part_range_round_trips_with_locate() { - let s = PartSchedule::default(); - // Spot-check a few part indices in different tiers. - for &idx in &[0u32, 1, 999, 1000, 1500, 1999, 2000, 5000, 9999] { - let r = s.part_range(idx).unwrap(); - let loc = s.locate(r.start).unwrap(); - assert_eq!(loc.part_index, idx, "round-trip idx {idx}"); - assert_eq!(loc.part_start, r.start); - assert_eq!(loc.part_size, r.end - r.start); - } - assert!(s.part_range(10_000).is_none()); + let c = Config::builder().bucket_prefix("").build(); + assert_eq!(c.store.prefix, "", "an empty prefix stays empty"); } #[test] - fn builder_overrides_defaults() { + fn builder_sets_every_field() { let c = Config::builder() - .memory_limit_bytes(8 * 1024 * 1024) - .max_parallel_parts(4) - .mount_path("/data") - .bucket_prefix("my/prefix") + .mount_path("/mnt") + .record_size(8192) + .max_symlink_depth(8) + .block_cache_bytes(1024) + .root_retention(None) .build(); - assert_eq!(c.memory_limit_bytes, 8 * 1024 * 1024); - assert_eq!(c.max_parallel_parts, 4); - assert_eq!(c.mount_path, "/data"); - assert_eq!(c.bucket_prefix, "my/prefix"); - // Unset fields keep defaults. - assert_eq!(c.max_symlink_depth, 40); + + assert_eq!(c.mount_path, "/mnt"); + assert_eq!(c.record_size(), 8192); + assert_eq!(c.max_symlink_depth, 8); + assert_eq!(c.store.block_cache_bytes, 1024); + assert_eq!(c.store.root_retention, None); + c.store.validate().unwrap(); } } diff --git a/crates/s3fs-core/src/crypto/aead.rs b/crates/s3fs-core/src/crypto/aead.rs new file mode 100644 index 0000000..594edd3 --- /dev/null +++ b/crates/s3fs-core/src/crypto/aead.rs @@ -0,0 +1,280 @@ +//! Authenticated encryption for stored blocks (AES-256-GCM). +//! +//! Two invariants carry the security of this module. Both are enforced by the +//! types rather than by convention, because getting either wrong is silent. +//! +//! **Nonces never repeat.** A nonce is `txg ‖ block_seq`, and a txg number is +//! consumed before a commit is attempted and never reused — not even when the +//! commit fails. Reusing one under a fixed key would leak the XOR of two +//! plaintexts and, worse for GCM, hand out the authentication subkey. +//! +//! **Every block is bound to its position.** The AAD covers the object id, +//! tree level, block index, and birth txg. A block that is genuine, correctly +//! encrypted, and correctly checksummed still fails to open if it is served +//! from a different offset, a different file, or a different level of the +//! indirect tree. Without this, a bucket operator could shuffle blocks between +//! positions and every individual integrity check would still pass. + +use aws_lc_rs::aead::{Aad as LcAad, LessSafeKey, Nonce, NONCE_LEN}; + +use crate::errors::{FsError, FsResult}; + +/// AEAD tag length for AES-256-GCM, in bytes. Sealed output is +/// `plaintext.len() + TAG_LEN`. +pub const TAG_LEN: usize = 16; + +/// Length of the encoded [`BlockAad`]. +const AAD_LEN: usize = 8 + 1 + 8 + 8; + +/// A 96-bit AES-GCM nonce, constructed only from a `(txg, block_seq)` pair. +/// +/// There is deliberately no way to build one from arbitrary bytes: every +/// nonce in the system is a function of a monotonically increasing txg, which +/// is what makes non-repetition an argument about txg allocation rather than +/// about every call site. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub struct BlockNonce([u8; NONCE_LEN]); + +impl BlockNonce { + /// `txg` (8 bytes, big-endian) followed by `block_seq` (4 bytes, + /// big-endian): the index of this block within its transaction group. + /// + /// Uniqueness argument: `txg` is strictly increasing across commits and + /// never reused, and `block_seq` is unique within a commit. So the pair is + /// unique over the lifetime of the key. + pub fn new(txg: u64, block_seq: u32) -> Self { + let mut n = [0u8; NONCE_LEN]; + n[..8].copy_from_slice(&txg.to_be_bytes()); + n[8..].copy_from_slice(&block_seq.to_be_bytes()); + BlockNonce(n) + } + + pub const fn as_bytes(&self) -> &[u8; NONCE_LEN] { + &self.0 + } + + pub const fn from_bytes(bytes: [u8; NONCE_LEN]) -> Self { + BlockNonce(bytes) + } + + fn to_lc(self) -> Nonce { + // Safe by the uniqueness argument above. + Nonce::assume_unique_for_key(self.0) + } +} + +/// Additional authenticated data binding a block to its position in the tree. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct BlockAad { + /// Object the block belongs to. + pub objid: u64, + /// 0 for data blocks, ≥1 for indirect blocks. + pub level: u8, + /// Index of this block within its level. + pub block_index: u64, + /// The txg that wrote it. + pub birth_txg: u64, +} + +impl BlockAad { + fn encode(&self) -> [u8; AAD_LEN] { + let mut b = [0u8; AAD_LEN]; + b[0..8].copy_from_slice(&self.objid.to_be_bytes()); + b[8] = self.level; + b[9..17].copy_from_slice(&self.block_index.to_be_bytes()); + b[17..25].copy_from_slice(&self.birth_txg.to_be_bytes()); + b + } +} + +/// Encrypt `plaintext`, returning `ciphertext ‖ tag`. +pub fn seal( + key: &LessSafeKey, + nonce: BlockNonce, + aad: BlockAad, + plaintext: &[u8], +) -> FsResult> { + let mut buf = Vec::with_capacity(plaintext.len() + TAG_LEN); + buf.extend_from_slice(plaintext); + key.seal_in_place_append_tag(nonce.to_lc(), LcAad::from(aad.encode()), &mut buf) + .map_err(|_| FsError::Integrity("block seal failed"))?; + Ok(buf) +} + +/// Decrypt `ciphertext ‖ tag` in place, returning the plaintext. +/// +/// Fails with [`FsError::Integrity`] if the tag does not verify — which covers +/// a corrupted block, a forged block, and a genuine block replayed at the +/// wrong position. +pub fn open( + key: &LessSafeKey, + nonce: BlockNonce, + aad: BlockAad, + ciphertext: &[u8], +) -> FsResult> { + if ciphertext.len() < TAG_LEN { + return Err(FsError::Integrity("block shorter than AEAD tag")); + } + let mut buf = ciphertext.to_vec(); + let len = key + .open_in_place(nonce.to_lc(), LcAad::from(aad.encode()), &mut buf) + .map_err(|_| FsError::Integrity("block authentication failed"))? + .len(); + buf.truncate(len); + Ok(buf) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::crypto::keys::{KeyMaterial, MasterSecret}; + + fn material(master: u8) -> KeyMaterial { + KeyMaterial::derive(&MasterSecret::from_bytes([master; 32]), [0u8; 16]).unwrap() + } + + fn key() -> KeyMaterial { + material(7) + } + + fn aad() -> BlockAad { + BlockAad { + objid: 42, + level: 0, + block_index: 3, + birth_txg: 100, + } + } + + #[test] + fn round_trip() { + let k = key(); + let pt = b"the quick brown fox".to_vec(); + let ct = seal(k.block_key(), BlockNonce::new(100, 0), aad(), &pt).unwrap(); + assert_eq!(ct.len(), pt.len() + TAG_LEN); + assert_ne!( + &ct[..pt.len()], + &pt[..], + "plaintext must not appear in output" + ); + + let out = open(k.block_key(), BlockNonce::new(100, 0), aad(), &ct).unwrap(); + assert_eq!(out, pt); + } + + #[test] + fn empty_plaintext_round_trips() { + let k = key(); + let ct = seal(k.block_key(), BlockNonce::new(1, 0), aad(), b"").unwrap(); + assert_eq!(ct.len(), TAG_LEN); + assert_eq!( + open(k.block_key(), BlockNonce::new(1, 0), aad(), &ct).unwrap(), + b"" + ); + } + + #[test] + fn tampered_ciphertext_fails() { + let k = key(); + let mut ct = seal(k.block_key(), BlockNonce::new(100, 0), aad(), b"payload").unwrap(); + ct[0] ^= 0x01; + assert!(matches!( + open(k.block_key(), BlockNonce::new(100, 0), aad(), &ct), + Err(FsError::Integrity(_)) + )); + } + + #[test] + fn tampered_tag_fails() { + let k = key(); + let mut ct = seal(k.block_key(), BlockNonce::new(100, 0), aad(), b"payload").unwrap(); + let last = ct.len() - 1; + ct[last] ^= 0x01; + assert!(open(k.block_key(), BlockNonce::new(100, 0), aad(), &ct).is_err()); + } + + #[test] + fn truncated_input_fails_cleanly() { + let k = key(); + assert!(matches!( + open( + k.block_key(), + BlockNonce::new(1, 0), + aad(), + &[0u8; TAG_LEN - 1] + ), + Err(FsError::Integrity("block shorter than AEAD tag")) + )); + } + + /// The position-binding property: a block that is genuine in every other + /// respect must not open at a different location in the tree. + #[test] + fn block_cannot_be_relocated() { + let k = key(); + let ct = seal(k.block_key(), BlockNonce::new(100, 0), aad(), b"payload").unwrap(); + + for wrong in [ + BlockAad { objid: 43, ..aad() }, + BlockAad { level: 1, ..aad() }, + BlockAad { + block_index: 4, + ..aad() + }, + BlockAad { + birth_txg: 101, + ..aad() + }, + ] { + assert!( + open(k.block_key(), BlockNonce::new(100, 0), wrong, &ct).is_err(), + "block opened under the wrong AAD: {wrong:?}" + ); + } + } + + #[test] + fn wrong_nonce_fails() { + let k = key(); + let ct = seal(k.block_key(), BlockNonce::new(100, 0), aad(), b"payload").unwrap(); + assert!(open(k.block_key(), BlockNonce::new(100, 1), aad(), &ct).is_err()); + assert!(open(k.block_key(), BlockNonce::new(101, 0), aad(), &ct).is_err()); + } + + #[test] + fn wrong_key_fails() { + let ct = seal( + key().block_key(), + BlockNonce::new(100, 0), + aad(), + b"payload", + ) + .unwrap(); + let other = material(8); + assert!(open(other.block_key(), BlockNonce::new(100, 0), aad(), &ct).is_err()); + } + + #[test] + fn nonce_layout_is_txg_then_seq() { + let n = BlockNonce::new(0x0102_0304_0506_0708, 0x090a_0b0c); + assert_eq!( + n.as_bytes(), + &[0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0a, 0x0b, 0x0c] + ); + } + + /// The uniqueness argument reduced to a property test: distinct + /// `(txg, seq)` pairs must produce distinct nonces. + #[test] + fn distinct_txg_seq_pairs_give_distinct_nonces() { + let mut seen = std::collections::HashSet::new(); + for txg in 0..64u64 { + for seq in 0..64u32 { + assert!( + seen.insert(*BlockNonce::new(txg, seq).as_bytes()), + "nonce collision at txg={txg} seq={seq}" + ); + } + } + } +} diff --git a/crates/s3fs-core/src/crypto/hash.rs b/crates/s3fs-core/src/crypto/hash.rs new file mode 100644 index 0000000..def5a63 --- /dev/null +++ b/crates/s3fs-core/src/crypto/hash.rs @@ -0,0 +1,189 @@ +//! BLAKE3 hashing — the Merkle-tree primitive. +//! +//! Every block pointer carries a [`Hash256`] of the *stored* (post-encryption) +//! bytes of the block it points at. That is what makes the tree a Merkle tree: +//! a parent authenticates each child independently of the AEAD key, so +//! integrity is checkable by a party that cannot decrypt, and a tampered block +//! is caught before it is ever handed to the cipher. + +use std::fmt; + +use crate::errors::{FsError, FsResult}; + +/// Length of a hash in bytes. +pub const HASH_LEN: usize = 32; + +/// A BLAKE3-256 digest. +#[derive(Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)] +pub struct Hash256([u8; HASH_LEN]); + +impl Hash256 { + /// The all-zero hash. Used as the checksum of a hole (a + /// [`crate::store::BlkPtr`] that points at nothing) and as the + /// `prev_root_hash` of the genesis root record. + pub const ZERO: Hash256 = Hash256([0u8; HASH_LEN]); + + /// Hash arbitrary bytes. + pub fn of(bytes: &[u8]) -> Self { + Hash256(*blake3::hash(bytes).as_bytes()) + } + + /// Keyed hash. Used for directory name hashing, where an unkeyed hash + /// would let anyone who can see the bucket craft names that collide into + /// one bucket block. + pub fn keyed(key: &[u8; 32], bytes: &[u8]) -> Self { + Hash256(*blake3::keyed_hash(key, bytes).as_bytes()) + } + + pub const fn from_bytes(bytes: [u8; HASH_LEN]) -> Self { + Hash256(bytes) + } + + pub const fn as_bytes(&self) -> &[u8; HASH_LEN] { + &self.0 + } + + pub fn is_zero(&self) -> bool { + self.0 == [0u8; HASH_LEN] + } + + /// First 8 bytes as a `u64`, for hash-bucket indexing. Big-endian so that + /// "the top `d` bits" is a prefix of the hex representation, which makes + /// extendible-hash bucket indices legible in logs and test failures. + pub fn prefix_u64(&self) -> u64 { + u64::from_be_bytes(self.0[..8].try_into().expect("32 >= 8")) + } + + /// Constant-time comparison against an expected value. + /// + /// Verification results are not secret, but a data-dependent early exit + /// here would leak how many leading bytes of a forged block matched, which + /// is enough to mount a byte-at-a-time forgery search against any code + /// path that lets an attacker submit candidate blocks. + pub fn verify(&self, expected: &Hash256, context: &'static str) -> FsResult<()> { + let mut diff = 0u8; + for i in 0..HASH_LEN { + diff |= self.0[i] ^ expected.0[i]; + } + if diff == 0 { + Ok(()) + } else { + Err(FsError::Integrity(context)) + } + } +} + +/// Defaults to [`Hash256::ZERO`], which is the checksum of a hole. This is +/// what makes `BlkPtr::default()` a valid hole rather than a nonsense pointer. +impl Default for Hash256 { + fn default() -> Self { + Hash256::ZERO + } +} + +impl fmt::Debug for Hash256 { + /// Abbreviated so log lines and assertion failures stay readable. Eight + /// hex characters is enough to tell two blocks apart when debugging and + /// short enough not to swamp a trace. + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!( + f, + "{:02x}{:02x}{:02x}{:02x}…", + self.0[0], self.0[1], self.0[2], self.0[3] + ) + } +} + +impl fmt::Display for Hash256 { + /// Full lowercase hex. + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + for b in &self.0 { + write!(f, "{b:02x}")?; + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn known_answer_matches_blake3_spec() { + // BLAKE3 of the empty input, from the reference test vectors. + assert_eq!( + Hash256::of(b"").to_string(), + "af1349b9f5f9a1a6a0404dea36dcc9499bcb25c9adc112b7cc9a93cae41f3262" + ); + // BLAKE3 of "abc". + assert_eq!( + Hash256::of(b"abc").to_string(), + "6437b3ac38465133ffb63b75273a8db548c558465d79db03fd359c6cd5bd9d85" + ); + } + + #[test] + fn zero_is_not_the_hash_of_anything_we_write() { + // A hole is encoded as an all-zero checksum. If that collided with a + // real block's hash, a hole and a block would be indistinguishable. + assert!(Hash256::ZERO.is_zero()); + assert!(!Hash256::of(b"").is_zero()); + assert!(!Hash256::of(&[0u8; 4096]).is_zero()); + } + + #[test] + fn keying_changes_the_digest() { + let k1 = [1u8; 32]; + let k2 = [2u8; 32]; + assert_ne!(Hash256::keyed(&k1, b"name"), Hash256::keyed(&k2, b"name")); + assert_ne!(Hash256::keyed(&k1, b"name"), Hash256::of(b"name")); + // Deterministic for a fixed key. + assert_eq!(Hash256::keyed(&k1, b"name"), Hash256::keyed(&k1, b"name")); + } + + #[test] + fn verify_accepts_match_and_rejects_mismatch() { + let h = Hash256::of(b"block"); + assert!(h.verify(&h, "test").is_ok()); + + let other = Hash256::of(b"block!"); + assert!(matches!( + h.verify(&other, "blkptr checksum"), + Err(FsError::Integrity("blkptr checksum")) + )); + } + + #[test] + fn verify_rejects_a_single_flipped_bit() { + let h = Hash256::of(b"block"); + let mut tampered = *h.as_bytes(); + tampered[31] ^= 0x01; + assert!(Hash256::from_bytes(tampered).verify(&h, "test").is_err()); + + let mut tampered = *h.as_bytes(); + tampered[0] ^= 0x80; + assert!(Hash256::from_bytes(tampered).verify(&h, "test").is_err()); + } + + #[test] + fn round_trips_through_bytes() { + let h = Hash256::of(b"round trip"); + assert_eq!(Hash256::from_bytes(*h.as_bytes()), h); + } + + #[test] + fn prefix_is_the_big_endian_leading_bytes() { + let h = Hash256::from_bytes([ + 0x01, 0x23, 0x45, 0x67, 0x89, 0xab, 0xcd, 0xef, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + ]); + assert_eq!(h.prefix_u64(), 0x0123_4567_89ab_cdef); + } + + #[test] + fn debug_is_short_display_is_full() { + let h = Hash256::of(b"abc"); + assert_eq!(format!("{h:?}"), "6437b3ac…"); + assert_eq!(h.to_string().len(), 64); + } +} diff --git a/crates/s3fs-core/src/crypto/keys.rs b/crates/s3fs-core/src/crypto/keys.rs new file mode 100644 index 0000000..a3857eb --- /dev/null +++ b/crates/s3fs-core/src/crypto/keys.rs @@ -0,0 +1,303 @@ +//! Key hierarchy. +//! +//! Everything hangs off one 32-byte [`MasterSecret`]. In production that +//! secret comes from `kms:Decrypt` with a `Recipient` attestation document, so +//! KMS releases it only to an enclave whose PCRs match the key policy; in +//! development and tests it comes from config. The derivation below is +//! identical either way, which is the point: the on-disk format does not +//! change when attestation lands. +//! +//! ```text +//! MasterSecret (32 bytes, never leaves memory) +//! │ +//! HKDF-SHA384(salt = fs_uuid, info = purpose) +//! ┌──────────────────┼──────────────────┐ +//! ▼ ▼ ▼ +//! block AEAD key Ed25519 seed dir-hash key +//! (AES-256-GCM) (root signing) (keyed BLAKE3) +//! ``` +//! +//! Separate keys per purpose so that a weakness in one use cannot be pivoted +//! into another — in particular, the directory hash key is exposed to +//! chosen-input attacks (a guest picks filenames) in a way the block key +//! never is. + +use aws_lc_rs::aead::{LessSafeKey, UnboundKey, AES_256_GCM}; +use aws_lc_rs::hkdf::{KeyType, Salt, HKDF_SHA384}; +use aws_lc_rs::signature::{Ed25519KeyPair, KeyPair}; +use zeroize::{Zeroize, ZeroizeOnDrop}; + +use crate::errors::{FsError, FsResult}; + +/// HKDF info strings. Versioned so a future format revision can rotate +/// derived keys without changing the master secret. +const INFO_BLOCK: &[u8] = b"s3fs/block/v1"; +const INFO_ROOTSIGN: &[u8] = b"s3fs/rootsign/v1"; +const INFO_DIRHASH: &[u8] = b"s3fs/dirhash/v1"; +/// Seals the runtime's own secrets — the ACME account key and the TLS +/// certificate's private key — which live outside the filesystem. +/// +/// Outside, because a tenant's guest is confined to its own directory but the +/// runtime's own material must be unreachable from *any* of them. A separate +/// label keeps it cryptographically distinct from block data as well as +/// physically separate. +const INFO_RUNTIME_SEAL: &[u8] = b"s3fs/runtime-seal/v1"; + +/// Length of an Ed25519 public key and of a raw signing seed. +pub const ED25519_PUBLIC_KEY_LEN: usize = 32; +/// Length of an Ed25519 signature. +pub const ED25519_SIGNATURE_LEN: usize = 64; + +/// The root of the key hierarchy. Scrubbed on drop. +#[derive(Clone, Zeroize, ZeroizeOnDrop)] +pub struct MasterSecret([u8; 32]); + +impl MasterSecret { + pub const fn from_bytes(bytes: [u8; 32]) -> Self { + MasterSecret(bytes) + } + + /// The raw bytes. + /// + /// Named to be conspicuous at the call site. There is exactly one + /// legitimate reason to reach in here — sealing the secret so it can be + /// stored — and everything else should take a `&MasterSecret` and derive + /// what it needs through [`KeyMaterial`], which is why no plain `as_bytes` + /// exists. + pub fn expose_secret(&self) -> &[u8; 32] { + &self.0 + } + + /// Parse a 64-character hex string, as supplied by `--data-key` or an + /// environment variable in development. + pub fn from_hex(s: &str) -> FsResult { + let s = s.trim(); + if s.len() != 64 { + return Err(FsError::Invalid("master key must be 64 hex characters")); + } + let mut out = [0u8; 32]; + for (i, byte) in out.iter_mut().enumerate() { + *byte = u8::from_str_radix(&s[i * 2..i * 2 + 2], 16) + .map_err(|_| FsError::Invalid("master key is not valid hex"))?; + } + Ok(MasterSecret(out)) + } + + /// Generate a fresh random secret. Used when formatting a new filesystem + /// in development; production keys come from KMS. + pub fn generate() -> FsResult { + let mut out = [0u8; 32]; + aws_lc_rs::rand::fill(&mut out) + .map_err(|_| FsError::Io("system RNG unavailable".to_string()))?; + Ok(MasterSecret(out)) + } +} + +/// Deliberately opaque: a `Debug` that printed the bytes would leak the key +/// into any log line that formats a struct containing one. +impl std::fmt::Debug for MasterSecret { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str("MasterSecret()") + } +} + +/// Arbitrary-length HKDF output. +struct OkmLen(usize); +impl KeyType for OkmLen { + fn len(&self) -> usize { + self.0 + } +} + +fn hkdf(secret: &MasterSecret, salt: &[u8], info: &[u8], out: &mut [u8]) -> FsResult<()> { + let prk = Salt::new(HKDF_SHA384, salt).extract(&secret.0); + prk.expand(&[info], OkmLen(out.len())) + .and_then(|okm| okm.fill(out)) + .map_err(|_| FsError::Io("HKDF expansion failed".to_string())) +} + +/// All keys derived from one master secret, for one filesystem. +pub struct KeyMaterial { + block_key: LessSafeKey, + runtime_seal_key: LessSafeKey, + signing_key: Ed25519KeyPair, + public_key: [u8; ED25519_PUBLIC_KEY_LEN], + dir_hash_key: [u8; 32], + fs_uuid: [u8; 16], +} + +impl KeyMaterial { + /// Derive every per-purpose key. + /// + /// `fs_uuid` is the HKDF salt, so two filesystems formatted from the same + /// master secret still get independent keys. It is stored in the root + /// record, and is not a secret. + pub fn derive(master: &MasterSecret, fs_uuid: [u8; 16]) -> FsResult { + let mut block_bytes = [0u8; 32]; + hkdf(master, &fs_uuid, INFO_BLOCK, &mut block_bytes)?; + let block_key = LessSafeKey::new( + UnboundKey::new(&AES_256_GCM, &block_bytes) + .map_err(|_| FsError::Io("AES key setup failed".to_string()))?, + ); + block_bytes.zeroize(); + + let mut seal_bytes = [0u8; 32]; + hkdf(master, &fs_uuid, INFO_RUNTIME_SEAL, &mut seal_bytes)?; + let runtime_seal_key = LessSafeKey::new( + UnboundKey::new(&AES_256_GCM, &seal_bytes) + .map_err(|_| FsError::Io("AES key setup failed".to_string()))?, + ); + seal_bytes.zeroize(); + + let mut seed = [0u8; 32]; + hkdf(master, &fs_uuid, INFO_ROOTSIGN, &mut seed)?; + let signing_key = Ed25519KeyPair::from_seed_unchecked(&seed) + .map_err(|_| FsError::Io("Ed25519 key derivation failed".to_string()))?; + seed.zeroize(); + + let mut public_key = [0u8; ED25519_PUBLIC_KEY_LEN]; + public_key.copy_from_slice(signing_key.public_key().as_ref()); + + let mut dir_hash_key = [0u8; 32]; + hkdf(master, &fs_uuid, INFO_DIRHASH, &mut dir_hash_key)?; + + Ok(KeyMaterial { + block_key, + runtime_seal_key, + signing_key, + public_key, + dir_hash_key, + fs_uuid, + }) + } + + pub fn block_key(&self) -> &LessSafeKey { + &self.block_key + } + + /// Seals runtime secrets stored outside the filesystem. See + /// [`INFO_RUNTIME_SEAL`]. + pub fn runtime_seal_key(&self) -> &LessSafeKey { + &self.runtime_seal_key + } + + pub fn signing_key(&self) -> &Ed25519KeyPair { + &self.signing_key + } + + /// Public half of the root-signing key. Written into each root record so + /// an external auditor can verify the chain without the master secret. + pub fn public_key(&self) -> &[u8; ED25519_PUBLIC_KEY_LEN] { + &self.public_key + } + + pub fn dir_hash_key(&self) -> &[u8; 32] { + &self.dir_hash_key + } + + pub fn fs_uuid(&self) -> &[u8; 16] { + &self.fs_uuid + } +} + +impl Drop for KeyMaterial { + fn drop(&mut self) { + self.dir_hash_key.zeroize(); + } +} + +impl std::fmt::Debug for KeyMaterial { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("KeyMaterial") + .field("fs_uuid", &hex16(&self.fs_uuid)) + .field("public_key", &hex16(&self.public_key[..8])) + .finish_non_exhaustive() + } +} + +fn hex16(bytes: &[u8]) -> String { + bytes.iter().map(|b| format!("{b:02x}")).collect() +} + +#[cfg(test)] +mod tests { + use super::*; + + fn master() -> MasterSecret { + MasterSecret::from_bytes([0x5a; 32]) + } + + #[test] + fn derivation_is_deterministic() { + let a = KeyMaterial::derive(&master(), [1u8; 16]).unwrap(); + let b = KeyMaterial::derive(&master(), [1u8; 16]).unwrap(); + assert_eq!(a.public_key(), b.public_key()); + assert_eq!(a.dir_hash_key(), b.dir_hash_key()); + } + + /// The salt is what keeps two filesystems formatted from the same master + /// secret cryptographically separate. + #[test] + fn different_fs_uuid_gives_different_keys() { + let a = KeyMaterial::derive(&master(), [1u8; 16]).unwrap(); + let b = KeyMaterial::derive(&master(), [2u8; 16]).unwrap(); + assert_ne!(a.public_key(), b.public_key()); + assert_ne!(a.dir_hash_key(), b.dir_hash_key()); + } + + #[test] + fn different_master_gives_different_keys() { + let a = KeyMaterial::derive(&MasterSecret::from_bytes([1u8; 32]), [0u8; 16]).unwrap(); + let b = KeyMaterial::derive(&MasterSecret::from_bytes([2u8; 32]), [0u8; 16]).unwrap(); + assert_ne!(a.public_key(), b.public_key()); + assert_ne!(a.dir_hash_key(), b.dir_hash_key()); + } + + /// Purpose separation: the signing seed and the directory hash key are + /// derived from the same secret and salt, differing only in the info + /// string. If those collided, the info strings would not be doing + /// anything. + #[test] + fn purposes_are_separated() { + let km = KeyMaterial::derive(&master(), [0u8; 16]).unwrap(); + assert_ne!(&km.dir_hash_key()[..], &km.public_key()[..]); + } + + #[test] + fn hex_parsing_round_trips() { + let hex = "0102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f20"; + let m = MasterSecret::from_hex(hex).unwrap(); + assert_eq!(m.0[0], 0x01); + assert_eq!(m.0[31], 0x20); + // Whitespace from a config file or shell is tolerated. + assert!(MasterSecret::from_hex(&format!(" {hex}\n")).is_ok()); + } + + #[test] + fn hex_parsing_rejects_bad_input() { + assert!(MasterSecret::from_hex("abcd").is_err()); + assert!(MasterSecret::from_hex(&"z".repeat(64)).is_err()); + assert!(MasterSecret::from_hex("").is_err()); + } + + #[test] + fn generate_produces_distinct_secrets() { + let a = MasterSecret::generate().unwrap(); + let b = MasterSecret::generate().unwrap(); + assert_ne!(a.0, b.0); + assert_ne!(a.0, [0u8; 32], "RNG returned all zeros"); + } + + /// A key that prints itself is a key in the logs. + #[test] + fn secrets_are_redacted_in_debug_output() { + let m = MasterSecret::from_bytes([0xab; 32]); + let s = format!("{m:?}"); + assert_eq!(s, "MasterSecret()"); + assert!(!s.contains("ab")); + + let km = KeyMaterial::derive(&m, [0u8; 16]).unwrap(); + let s = format!("{km:?}"); + assert!(!s.contains(&hex16(km.dir_hash_key()))); + } +} diff --git a/crates/s3fs-core/src/crypto/mod.rs b/crates/s3fs-core/src/crypto/mod.rs new file mode 100644 index 0000000..3181955 --- /dev/null +++ b/crates/s3fs-core/src/crypto/mod.rs @@ -0,0 +1,25 @@ +//! Cryptographic primitives for the block store. +//! +//! Three jobs, three submodules: +//! +//! - [`hash`] — BLAKE3 digests. Every block pointer carries the hash of the +//! block it points at, which is what makes the object tree a Merkle tree. +//! - [`aead`] — AES-256-GCM over stored blocks, with nonces derived from the +//! txg and additional data binding each block to its position in the tree. +//! - [`sign`] — Ed25519 over root records, so a root cannot be minted or +//! re-pointed by anyone holding only bucket write access. +//! +//! [`keys`] ties them together: one master secret, HKDF-separated per purpose. +//! +//! Primitives come from `aws-lc-rs` (the same stack KMS speaks, FIPS-capable) +//! except BLAKE3, which has no aws-lc equivalent and is chosen for the block +//! checksum because it is the hot path — every block read verifies one. + +pub mod aead; +pub mod hash; +pub mod keys; +pub mod sign; + +pub use aead::{BlockAad, BlockNonce, TAG_LEN}; +pub use hash::{Hash256, HASH_LEN}; +pub use keys::{KeyMaterial, MasterSecret, ED25519_PUBLIC_KEY_LEN, ED25519_SIGNATURE_LEN}; diff --git a/crates/s3fs-core/src/crypto/sign.rs b/crates/s3fs-core/src/crypto/sign.rs new file mode 100644 index 0000000..d574a31 --- /dev/null +++ b/crates/s3fs-core/src/crypto/sign.rs @@ -0,0 +1,110 @@ +//! Ed25519 signatures over root records. +//! +//! The signature is what turns a pile of immutable objects into a filesystem +//! we are willing to trust. It covers the sequence number and the previous +//! root's hash as well as the Merkle root, so the *chain* is authenticated, +//! not merely each link: an attacker who can write to the bucket cannot mint a +//! root, cannot re-point an existing root at different data, and cannot splice +//! two legitimate histories together. + +use aws_lc_rs::signature::{Ed25519KeyPair, UnparsedPublicKey, ED25519}; + +use crate::errors::{FsError, FsResult}; + +use super::keys::{ED25519_PUBLIC_KEY_LEN, ED25519_SIGNATURE_LEN}; + +/// Sign `message`, returning a detached 64-byte signature. +pub fn sign(key: &Ed25519KeyPair, message: &[u8]) -> [u8; ED25519_SIGNATURE_LEN] { + let sig = key.sign(message); + let mut out = [0u8; ED25519_SIGNATURE_LEN]; + out.copy_from_slice(sig.as_ref()); + out +} + +/// Verify a detached signature. +/// +/// Returns [`FsError::Integrity`] on any failure. The caller must treat that +/// as fatal for the mount: a bad root signature means the store is serving +/// something we did not write. +pub fn verify( + public_key: &[u8; ED25519_PUBLIC_KEY_LEN], + message: &[u8], + signature: &[u8; ED25519_SIGNATURE_LEN], +) -> FsResult<()> { + UnparsedPublicKey::new(&ED25519, public_key) + .verify(message, signature) + .map_err(|_| FsError::Integrity("root signature verification failed")) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::crypto::keys::{KeyMaterial, MasterSecret}; + + fn material(master: u8) -> KeyMaterial { + KeyMaterial::derive(&MasterSecret::from_bytes([master; 32]), [0u8; 16]).unwrap() + } + + #[test] + fn sign_then_verify() { + let km = material(1); + let msg = b"root record bytes"; + let sig = sign(km.signing_key(), msg); + assert!(verify(km.public_key(), msg, &sig).is_ok()); + } + + #[test] + fn verify_rejects_modified_message() { + let km = material(1); + let sig = sign(km.signing_key(), b"root record bytes"); + assert!(matches!( + verify(km.public_key(), b"root record bytez", &sig), + Err(FsError::Integrity(_)) + )); + } + + #[test] + fn verify_rejects_modified_signature() { + let km = material(1); + let mut sig = sign(km.signing_key(), b"msg"); + sig[0] ^= 0x01; + assert!(verify(km.public_key(), b"msg", &sig).is_err()); + } + + /// The forgery case that matters: an attacker who controls the bucket can + /// write any bytes they like, but cannot produce a signature that our + /// public key accepts. + #[test] + fn verify_rejects_signature_from_another_key() { + let ours = material(1); + let theirs = material(2); + let sig = sign(theirs.signing_key(), b"forged root"); + assert!(verify(ours.public_key(), b"forged root", &sig).is_err()); + } + + #[test] + fn verify_rejects_all_zero_signature() { + let km = material(1); + assert!(verify(km.public_key(), b"msg", &[0u8; ED25519_SIGNATURE_LEN]).is_err()); + } + + #[test] + fn signatures_are_deterministic() { + // Ed25519 is deterministic; two signings of the same message with the + // same key must agree. Relied on by the commit path, which signs a + // canonical encoding and expects a stable root object. + let km = material(1); + assert_eq!( + sign(km.signing_key(), b"same message"), + sign(km.signing_key(), b"same message") + ); + } + + #[test] + fn empty_message_signs_and_verifies() { + let km = material(1); + let sig = sign(km.signing_key(), b""); + assert!(verify(km.public_key(), b"", &sig).is_ok()); + assert!(verify(km.public_key(), b"x", &sig).is_err()); + } +} diff --git a/crates/s3fs-core/src/errors.rs b/crates/s3fs-core/src/errors.rs index b4daafd..0157bd4 100644 --- a/crates/s3fs-core/src/errors.rs +++ b/crates/s3fs-core/src/errors.rs @@ -1,7 +1,7 @@ //! `FsError` — the engine's internal error type. //! //! Translation to WASI Preview 2 `wasi:filesystem/types::error-code` happens in -//! the `s3fs-wasmtime` crate. Inside the engine we keep error variants close to +//! the `enclave-runtime` crate. Inside the engine we keep error variants close to //! POSIX-flavored categories so call sites are obvious. use thiserror::Error; @@ -68,6 +68,34 @@ pub enum FsError { #[error("precondition / CAS conflict")] Conflict, + /// A block, dnode, or root record failed verification: checksum mismatch + /// against the parent block pointer, AEAD authentication failure, a bad + /// root signature, or a broken `prev_root_hash` chain link. + /// + /// This is never transient and must never be retried or degraded around. + /// Producing it poisons the mount: the store is either lying or corrupt, + /// and in both cases the only safe response is to stop serving bytes. + /// The store holds no filesystem. + /// + /// Distinct from `NotFound`, which is about a path inside a mounted + /// filesystem. This says the *store* is empty, and it is returned rather + /// than quietly formatting one: an enclave pointed at an empty store must + /// be able to tell "this filesystem is new" from "everything has been + /// hidden", and only the caller knows which it is entitled to assume. + #[error("the store holds no filesystem")] + NoFilesystem, + + #[error("integrity check failed: {0}")] + Integrity(&'static str), + + /// The store offered a root record older than one we have already + /// accepted, or older than the configured floor. Distinct from + /// [`FsError::Integrity`] because the data is *valid* — correctly signed + /// and internally consistent — just stale. That is the signature of a + /// rollback attempt rather than corruption. + #[error("rollback detected: root seq {found} is not newer than {expected}")] + Rollback { expected: u64, found: u64 }, + #[error("i/o timeout")] IoTimeout, @@ -108,6 +136,30 @@ mod tests { FsError::Invalid("bad part number").to_string(), "invalid argument: bad part number" ); + assert_eq!( + FsError::Integrity("blkptr checksum").to_string(), + "integrity check failed: blkptr checksum" + ); + assert_eq!( + FsError::Rollback { + expected: 42, + found: 41 + } + .to_string(), + "rollback detected: root seq 41 is not newer than 42" + ); + } + + /// Integrity and rollback failures are adversarial signals, not blips. + /// Retrying them would turn a detected attack into a spin loop. + #[test] + fn verification_failures_are_never_transient() { + assert!(!FsError::Integrity("root signature").is_transient()); + assert!(!FsError::Rollback { + expected: 2, + found: 1 + } + .is_transient()); } #[test] diff --git a/crates/s3fs-core/src/flusher.rs b/crates/s3fs-core/src/flusher.rs deleted file mode 100644 index 32aff47..0000000 --- a/crates/s3fs-core/src/flusher.rs +++ /dev/null @@ -1,281 +0,0 @@ -//! Background flusher: a long-running task that consumes `FlushJob`s from -//! an `mpsc` channel and runs them with bounded concurrency. -//! -//! Two job kinds today: -//! -//! - **`UploadPart`** — submitted eagerly by `pwrite` whenever a write -//! makes a part fully dirty. The worker spawns the upload (subject to -//! the per-`Fs` permit cap) without blocking the caller. -//! - **`Commit`** — submitted by `sync`. The worker drains every -//! in-flight `UploadPart` for the same handle, then runs -//! `mpu::commit` and replies via the supplied `oneshot`. -//! -//! Errors from background uploads are stashed on the [`FileHandle`] -//! (see `FileHandle::last_error`) and surfaced to the caller on the next -//! `pwrite` / `sync`. This keeps the user-facing async surface clean — -//! `pwrite` returns immediately even when its enqueued upload is yet to -//! run. - -use std::sync::Arc; - -use bytes::Bytes; -use tokio::sync::{mpsc, oneshot, OwnedSemaphorePermit, Semaphore}; -use tokio_util::sync::CancellationToken; - -use crate::backend::{Backend, BlobMeta, MultipartId, PutBlobInput}; -use crate::buffer::PartBuf; -use crate::config::{Config, PartSchedule}; -use crate::errors::{FsError, FsResult}; -use crate::mpu::{self, MpuState}; - -/// Outcome of an `UploadPart` job: the part index and its ETag. -pub type PartResult = FsResult<(u32, String)>; - -/// One unit of work for the background flusher. -pub enum FlushJob { - UploadPart { - backend: Arc, - key: String, - upload_id: MultipartId, - part_index: u32, - part_arc: Arc>, - schedule: PartSchedule, - source_size: u64, - final_size: u64, - /// One-shot to deliver the result back to the spawner. The - /// receiver typically lives on `PartBuf::upload_in_flight`. - reply: oneshot::Sender, - }, - /// Run a function with a permit acquired (used by `sync()` to issue - /// `UploadPartCopy` calls under the same global cap as `UploadPart`). - /// The body should NOT hold the permit longer than the actual S3 - /// call. - WithPermit { - body: Box tokio::task::JoinHandle<()> + Send>, - }, -} - -impl std::fmt::Debug for FlushJob { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - FlushJob::UploadPart { part_index, .. } => { - write!(f, "FlushJob::UploadPart {{ part_index: {part_index} }}") - } - FlushJob::WithPermit { .. } => write!(f, "FlushJob::WithPermit"), - } - } -} - -/// Per-`Fs` background flusher. -/// -/// Owns a long-running tokio task plus the channel + semaphore used to -/// serialise jobs to it. The worker runs each `UploadPart` job under one -/// permit from the semaphore so the total in-flight upload count across -/// all open files is bounded by `Config::max_parallel_parts`. -#[derive(Debug)] -pub struct Flusher { - job_tx: mpsc::UnboundedSender, - semaphore: Arc, - cancel: CancellationToken, - /// Kept alive so we can `.abort()` the worker on `drop`. - worker: parking_lot::Mutex>>, -} - -impl Flusher { - /// Spawn the background worker. Idempotent at the type level — call - /// once from `Fs::new`. - pub fn new(config: &Config) -> Self { - let (job_tx, mut job_rx) = mpsc::unbounded_channel::(); - let semaphore = Arc::new(Semaphore::new(config.max_parallel_parts.max(1))); - let cancel = CancellationToken::new(); - - let sem_for_worker = semaphore.clone(); - let cancel_for_worker = cancel.clone(); - let worker = tokio::spawn(async move { - loop { - tokio::select! { - biased; - _ = cancel_for_worker.cancelled() => break, - maybe_job = job_rx.recv() => { - let Some(job) = maybe_job else { break }; - match job { - FlushJob::UploadPart { backend, key, upload_id, part_index, part_arc, schedule, source_size, final_size, reply } => { - let sem = sem_for_worker.clone(); - tokio::spawn(async move { - let _permit = sem.acquire_owned().await.expect("semaphore never closed"); - let res = crate::fs::upload_one_part( - backend, &key, &upload_id, part_index, - part_arc, &schedule, source_size, final_size, - ).await; - let _ = reply.send(res); - }); - } - FlushJob::WithPermit { body } => { - let sem = sem_for_worker.clone(); - tokio::spawn(async move { - let permit = sem.acquire_owned().await.expect("semaphore never closed"); - let _ = body(permit).await; - }); - } - } - } - } - } - }); - - Self { - job_tx, - semaphore, - cancel, - worker: parking_lot::Mutex::new(Some(worker)), - } - } - - /// Submit an `UploadPart` job and return the receiver for its result. - /// The receiver should be parked on the part (so a later `sync` can - /// await it). If the channel is closed (worker dropped), returns an - /// error immediately. - #[allow(clippy::too_many_arguments)] - pub fn enqueue_upload( - &self, - backend: Arc, - key: String, - upload_id: MultipartId, - part_index: u32, - part_arc: Arc>, - schedule: PartSchedule, - source_size: u64, - final_size: u64, - ) -> Result, FsError> { - let (reply_tx, reply_rx) = oneshot::channel(); - self.job_tx - .send(FlushJob::UploadPart { - backend, - key, - upload_id, - part_index, - part_arc, - schedule, - source_size, - final_size, - reply: reply_tx, - }) - .map_err(|_| FsError::Io("flusher channel closed".into()))?; - Ok(reply_rx) - } - - /// Acquire a permit directly. Used by `mpu::commit`'s `UploadPartCopy` - /// fan-out so it shares the same global concurrency cap as - /// `UploadPart` calls. - pub async fn acquire(&self) -> OwnedSemaphorePermit { - self.semaphore - .clone() - .acquire_owned() - .await - .expect("semaphore never closed") - } - - /// Clone of the inner semaphore. Used by hot paths (e.g. - /// `upload_parts_concurrent`) that need to move it into spawned tasks - /// without cloning the whole `Flusher` (which owns the worker handle). - pub fn semaphore(&self) -> Arc { - self.semaphore.clone() - } - - pub fn available_permits(&self) -> usize { - self.semaphore.available_permits() - } -} - -impl Drop for Flusher { - fn drop(&mut self) { - // Signal cancellation; the worker checks this each loop iteration. - self.cancel.cancel(); - if let Some(handle) = self.worker.lock().take() { - handle.abort(); - } - } -} - -// --------------------------------------------------------------------------- -// Helpers used by `Fs::pwrite` and `Fs::sync`. -// --------------------------------------------------------------------------- - -/// Lazily begin an MPU on the handle if one isn't started already. -/// Returns the upload-id. Caller must hold the `MpuState` mutex. -pub async fn ensure_mpu_begun_locked( - backend: &Arc, - state: &mut Option, - key: &str, - schedule: &PartSchedule, - source_size: Option, - source_etag: Option, -) -> FsResult { - if let Some(s) = state.as_ref() { - if let Some(id) = &s.upload_id { - return Ok(id.clone()); - } - } - let id = backend - .multipart_begin(PutBlobInput { - key: key.to_string(), - body: Bytes::new(), - metadata: std::collections::HashMap::new(), - content_type: None, - }) - .await?; - if state.is_none() { - *state = Some(MpuState::new( - key.to_string(), - source_size, - source_etag, - schedule.clone(), - )); - } - state.as_mut().unwrap().upload_id = Some(id.clone()); - Ok(id) -} - -/// Run `mpu::commit` taking a permit from the flusher for the -/// CompleteMultipartUpload call. -pub async fn commit_via_flusher( - state: &mut MpuState, - backend: Arc, - max_merge_copy_bytes: u64, - max_parallel_copy: usize, - source_key: &str, -) -> FsResult { - mpu::commit( - state, - backend, - max_merge_copy_bytes, - max_parallel_copy, - source_key, - ) - .await -} - -#[cfg(test)] -mod tests { - use super::*; - - fn small_cfg(cap: usize) -> Config { - Config::builder().max_parallel_parts(cap).build() - } - - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn flusher_acquire_bounds_concurrency() { - let f = Flusher::new(&small_cfg(2)); - assert_eq!(f.available_permits(), 2); - let _p1 = f.acquire().await; - let _p2 = f.acquire().await; - assert_eq!(f.available_permits(), 0); - } - - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn flusher_drop_aborts_worker() { - let f = Flusher::new(&small_cfg(2)); - // No way to introspect from here, but Drop should not panic. - drop(f); - } -} diff --git a/crates/s3fs-core/src/fs.rs b/crates/s3fs-core/src/fs.rs index f20a4a0..7d577d9 100644 --- a/crates/s3fs-core/src/fs.rs +++ b/crates/s3fs-core/src/fs.rs @@ -1,65 +1,56 @@ -//! `Fs` — the public filesystem handle. +//! `Fs` — POSIX semantics over the block store. //! -//! Wires together the inode tree, the buffer pool, the MPU state machine, -//! and the backend into the API the WASI host adapter (and any other -//! consumer) calls into. Everything exposed here is async; locks are never -//! held across awaits. +//! Every mutation is one transaction group, and therefore one root record: a +//! `mkdir` either happened or did not, and there is no window in which half of +//! it is visible. That is a direct consequence of the store's commit protocol +//! rather than anything this layer arranges, and it is why several operations +//! that the previous engine could only approximate are now exact: //! -//! Scope of v1: -//! - `open`/`open_at` honoring `create`, `exclusive`, `truncate` -//! - `pread`/`pwrite` through `BufferPool` with on-demand S3 fetch -//! - `sync`: small-file fast path (single `PutObject`) or full MPU commit -//! (driver in [`crate::mpu`]) -//! - `mkdir`/`unlink`/`rmdir`/`rename` (synchronous file rename via -//! `CopyObject` + `DeleteObject` for now) -//! - `symlink_at`/`readlink_at` (storage convention only — follow-during- -//! lookup is reserved for the dedicated symlink module) +//! - **`rename` is atomic.** It moves a directory entry inside one commit. +//! There is no copy-then-delete window, no background queue, and a directory +//! rename costs the same as a file rename instead of `O(entries)` copies. +//! - **`stat` cannot go stale.** Attributes come from the dnode on every call. +//! The previous engine cached them behind a TTL that was never checked. +//! - **Identity is real.** `(objid, gen)` is stable across mounts. //! -//! Deferred: -//! - symlink resolution during `open_at` -//! - hardlink ops (`unsupported` per the Compatibility Matrix) +//! ## Buffering +//! +//! Writes accumulate in the handle as whole records and reach the store on +//! `sync`, `set_size`, or `close` — the transaction-group model the plan +//! specifies. Reads consult the handle's dirty records first, so a writer sees +//! its own unsynced writes; another handle does not, and on a crash they are +//! lost. What cannot happen is a torn or partially-applied state. -use std::collections::HashMap; +use std::collections::BTreeMap; use std::num::NonZeroU64; use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::Arc; +use std::time::SystemTime; -use bytes::{Bytes, BytesMut}; -use parking_lot::RwLock as PlRwLock; -use tokio::sync::Mutex as AsyncMutex; +use bytes::Bytes; +use parking_lot::Mutex as SyncMutex; +use std::collections::{HashMap, HashSet}; +use tokio::sync::Mutex; -use crate::backend::{Backend, BlobMeta, CopyBlobInput, PutBlobInput}; -use crate::buffer::{BufferPool, PartBuf, PartKey}; +use crate::backend::Backend; use crate::config::Config; +use crate::crypto::{KeyMaterial, MasterSecret}; use crate::errors::{FsError, FsResult}; -use crate::inode::{ - Attrs, DirEntry, Inode, InodeKind, InodeState, InodeTree, SYMLINK_METADATA_KEY, - SYMLINK_METADATA_VALUE, -}; -use crate::mpu::{self, MpuState}; -use crate::path; - -/// Wire-format `Content-Type` we attach to symlink objects so casual S3 -/// console viewers get a hint about what they are. -pub const SYMLINK_CONTENT_TYPE: &str = "application/x-s3wasifs-symlink"; - -/// User-metadata keys we use for `set-times`. Stored as RFC-3339-ish nanos. -pub const METADATA_ATIME_KEY: &str = "s3wasifs-atime"; -pub const METADATA_MTIME_KEY: &str = "s3wasifs-mtime"; - -fn systemtime_to_meta(t: std::time::SystemTime) -> String { - let d = t.duration_since(std::time::UNIX_EPOCH).unwrap_or_default(); - format!("{}.{:09}", d.as_secs(), d.subsec_nanos()) -} - -pub(crate) fn meta_to_systemtime(s: &str) -> Option { - let (sec_str, nanos_str) = s.split_once('.').unwrap_or((s, "0")); - let secs: u64 = sec_str.parse().ok()?; - let nanos: u32 = nanos_str.parse().ok()?; - Some(std::time::UNIX_EPOCH + std::time::Duration::new(secs, nanos)) -} - -/// Stable handle ID for an open file. Allocated by the `Fs` instance. +use crate::inode::{now_nanos, to_nanos, Attrs, DirEntry, Inode, InodeKind}; +use crate::path::{validate_segment, validate_symlink_target}; +use crate::store::blockstore::BlockStore; +use crate::store::dir::{DirTxn, Dirent}; +use crate::store::dnode::{blocks_for_size, Dnode, DnodeKind, INLINE_CAP, ROOT_OBJID}; +use crate::store::indirect::{block_logical_len, commit_object, read_data_block}; +use crate::store::objset::ObjectSet; +use crate::store::Store; + +/// Default permissions for objects this layer creates. +const DEFAULT_FILE_MODE: u32 = 0o100644; +const DEFAULT_DIR_MODE: u32 = 0o040755; +const DEFAULT_SYMLINK_MODE: u32 = 0o120777; + +/// Stable handle ID for an open file. #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)] pub struct HandleId(pub NonZeroU64); @@ -111,2516 +102,1449 @@ impl OpenFlags { } } -/// Open-file state. Multiple handles for the same inode have independent -/// MPU state (that's fine — last writer's `sync` wins). +/// Unsynced state of one open file. +#[derive(Debug)] +struct HandleState { + /// Size as this handle sees it, including unsynced growth. + size: u64, + /// Whole records awaiting commit, keyed by record index. + dirty: BTreeMap>, + /// Timestamps or size changed without any data changing. + meta_dirty: bool, + atime_nanos: u64, + mtime_nanos: u64, +} + +/// An open file. #[derive(Debug)] pub struct FileHandle { pub id: HandleId, pub inode: Arc, pub flags: OpenFlags, - /// Per-handle MPU bookkeeping. `None` until the first dirty write makes - /// us decide whether to begin an MPU. - pub mpu: AsyncMutex>, - /// File size as observed by this handle. Diverges from `inode.attrs.size` - /// during writes; reconciled by `sync`. - pub size: PlRwLock, + state: Mutex, } impl FileHandle { - fn new(id: HandleId, inode: Arc, flags: OpenFlags) -> Self { - let size = inode.attrs.read().size; - Self { - id, - inode, - flags, - mpu: AsyncMutex::new(None), - size: PlRwLock::new(size), + /// Size including writes not yet committed. + pub async fn size(&self) -> u64 { + self.state.lock().await.size + } +} + +/// Objects that have lost their last name but are still open. +/// +/// POSIX keeps an unlinked file alive until the last descriptor closes. The +/// dnode stays allocated with `nlink == 0`, unreachable by name — path +/// resolution goes through directory entries, and there are none left — so the +/// only way to it is a handle that already existed. When the last one closes, +/// it is freed. +/// +/// A crash in that window leaks the dnode: allocated, nameless, unreachable. +/// That is garbage, not corruption, and it is the same trade a real filesystem +/// makes with its orphan inode list. +#[derive(Debug, Default)] +struct OpenTracker { + /// Open handle count per object id. + counts: HashMap, + /// Objects awaiting their last close. + orphans: HashSet, +} + +impl OpenTracker { + fn opened(&mut self, objid: u64) { + *self.counts.entry(objid).or_insert(0) += 1; + } + + /// Record a close. Returns `true` if this was the last handle on an + /// orphaned object, meaning the caller must now free it. + fn closed(&mut self, objid: u64) -> bool { + let remaining = match self.counts.get_mut(&objid) { + Some(n) => { + *n = n.saturating_sub(1); + *n + } + None => 0, + }; + if remaining == 0 { + self.counts.remove(&objid); + return self.orphans.remove(&objid); } + false + } + + fn is_open(&self, objid: u64) -> bool { + self.counts.contains_key(&objid) + } + + fn mark_orphan(&mut self, objid: u64) { + self.orphans.insert(objid); } } -/// Public filesystem handle. Cheap to clone (`Arc`). +/// The filesystem. #[derive(Debug)] pub struct Fs { - pub backend: Arc, + store: Arc, pub config: Arc, - pub tree: Arc, - pub pool: Arc, - pub flusher: crate::flusher::Flusher, - pub rename_queue: crate::rename::RenameQueue, - handles: PlRwLock>>, + handles: SyncMutex>>, + open: SyncMutex, next_handle: AtomicU64, } impl Fs { - /// Construct a fresh `Fs` over `backend` with `config`. - pub fn new(backend: Arc, config: Arc) -> Arc { - let tree = InodeTree::new(backend.clone(), config.clone()); - let pool = BufferPool::new(config.clone()); - let flusher = crate::flusher::Flusher::new(&config); - let rename_queue = crate::rename::RenameQueue::new(); - Arc::new(Self { - backend, + /// Mount an existing filesystem. Fails with [`FsError::NoFilesystem`] if + /// the store is empty — see [`Store::open_existing`] for why that is not a + /// cue to create one. + /// + /// `roots` is the Object Lock bucket holding the anchor chain; `data` + /// holds the slabs. They may be the same bucket, but splitting them is + /// what lets dead copy-on-write blocks stay reclaimable while the anchor + /// stays immutable. + /// + /// `fs_uuid` is the HKDF salt. It is not a secret, but it must be supplied + /// rather than discovered: the keys that verify a root record are derived + /// from it, so reading it out of the store would mean trusting the store + /// to tell us which key to check its own signature with. + pub async fn mount( + data: Arc, + roots: Arc, + master: &MasterSecret, + fs_uuid: [u8; 16], + config: Arc, + min_root_seq: Option, + ) -> FsResult> { + let keys = Arc::new(KeyMaterial::derive(master, fs_uuid)?); + let store = Store::open_existing( + data, + roots, + keys, + Arc::new(config.store.clone()), + min_root_seq, + ) + .await?; + Ok(Fs::from_store(Arc::new(store), config)) + } + + /// Create a filesystem in an empty store. + /// + /// Separate from [`Fs::mount`] on purpose: creating one is an assertion + /// that no filesystem should already exist here, and the caller is the only + /// party that can make it. + pub async fn create( + data: Arc, + roots: Arc, + master: &MasterSecret, + fs_uuid: [u8; 16], + config: Arc, + ) -> FsResult> { + let keys = Arc::new(KeyMaterial::derive(master, fs_uuid)?); + let store = Store::create(data, roots, keys, Arc::new(config.store.clone())).await?; + Ok(Fs::from_store(Arc::new(store), config)) + } + + pub fn from_store(store: Arc, config: Arc) -> Arc { + Arc::new(Fs { + store, config, - tree, - pool, - flusher, - rename_queue, - handles: PlRwLock::new(HashMap::new()), + handles: SyncMutex::new(HashMap::new()), + open: SyncMutex::new(OpenTracker::default()), next_handle: AtomicU64::new(1), }) } - pub fn root(&self) -> Arc { - self.tree.root() + pub fn store(&self) -> &Arc { + &self.store } - fn alloc_handle_id(&self) -> HandleId { - let n = self.next_handle.fetch_add(1, Ordering::Relaxed); - HandleId(NonZeroU64::new(n).expect("handle counter overflow")) + /// The mount's root directory. + pub fn root(&self) -> Arc { + Inode::new(ROOT_OBJID, 0) } pub fn get_handle(&self, id: HandleId) -> Option> { - self.handles.read().get(&id).cloned() + self.handles.lock().get(&id.get()).cloned() } - // ------------------- read-only navigation ------------------- - - /// Stat an inode (pure cached read; no S3 call). - pub fn stat(&self, ino: &Arc) -> Attrs { - ino.attrs.read().clone() + fn blocks(&self) -> Arc { + self.store.blocks().clone() } - /// Resolve `path` from `base` (openat semantics) and stat the result. - pub async fn stat_at(&self, base: &Arc, path: &str) -> FsResult { - let ino = self.tree.lookup_at(base, path).await?; - Ok(self.stat(&ino)) + async fn objset(&self) -> FsResult { + self.store.objset().await } - /// Snapshot a directory's contents. - pub async fn read_dir(&self, dir: &Arc) -> FsResult> { - self.tree.snapshot_directory(dir).await + async fn dnode(&self, objid: u64) -> FsResult { + self.objset() + .await? + .get_allocated(&self.blocks(), objid) + .await } - // ------------------- mutation ------------------- + // -- resolution --------------------------------------------------------- - /// `mkdir` — write an explicit zero-byte directory marker. - /// Returns the new directory inode. - pub async fn mkdir(&self, base: &Arc, name: &str) -> FsResult> { - if !base.is_dir() { - return Err(FsError::NotDirectory); - } - if base.is_deleted() { - return Err(FsError::NotFound); - } - path::validate_segment(name)?; + /// Resolve `path` relative to `base`, following symlinks except optionally + /// the final component. + pub async fn lookup_at(&self, base: &Arc, path: &str) -> FsResult> { + self.resolve(base, path, true).await + } - // Reject if a child of the same name exists. - if self.tree.lookup(base, name).await.is_ok() { - return Err(FsError::AlreadyExists); - } + pub async fn lookup_at_no_follow(&self, base: &Arc, path: &str) -> FsResult> { + self.resolve(base, path, false).await + } - let parent_key = self.tree.s3_key(base); - let dir_key = if parent_key.is_empty() { - format!("{name}/") - } else { - format!("{parent_key}/{name}/") - }; - let meta = self - .backend - .put_blob(PutBlobInput { - key: dir_key.clone(), - body: Bytes::new(), - metadata: HashMap::new(), - content_type: None, - }) - .await?; + async fn resolve( + &self, + base: &Arc, + path: &str, + follow_final: bool, + ) -> FsResult> { + self.resolve_within(ROOT_OBJID, base, path, follow_final) + .await + } - let attrs = attrs_from_meta(&meta); - let id = self.tree.alloc_id(); - let inode = Inode::new_dir(id, name, Some(Arc::downgrade(base)), true, attrs); - Ok(self.tree.attach(base, inode)) + /// Resolve without leaving the subtree rooted at `scope`. + /// + /// This is the capability boundary between two tenants of one filesystem. + /// See [`resolve_in`] for what it forecloses and why those are the only + /// three routes out. + async fn resolve_within( + &self, + scope: u64, + base: &Arc, + path: &str, + follow_final: bool, + ) -> FsResult> { + let objset = self.objset().await?; + let blocks = self.blocks(); + let objid = resolve_in( + &objset, + &blocks, + scope, + base.objid(), + path, + follow_final, + self.config.max_symlink_depth, + ) + .await?; + let d = objset.get_allocated(&blocks, objid).await?; + Ok(Inode::new(objid, d.gen)) } - /// `unlink` — delete a regular file or symlink. - pub async fn unlink(&self, base: &Arc, name: &str) -> FsResult<()> { - let ino = self.tree.lookup(base, name).await?; - if ino.is_dir() { - return Err(FsError::IsDirectory); - } - // If the inode is mid-rename, wait for it to finish so we don't race - // the worker (which would otherwise resurrect a key we just deleted). - self.wait_for_rename(&ino).await?; - let key = self.tree.current_s3_key(&ino); - self.backend.delete_blob(&key).await?; - self.tree.detach(&ino); - // Drop any cached parts. - self.pool.forget_inode(ino.id); - Ok(()) + /// [`Fs::lookup_at`], confined to `scope`. + pub async fn lookup_within( + &self, + scope: &Arc, + base: &Arc, + path: &str, + ) -> FsResult> { + self.resolve_within(scope.objid(), base, path, true).await } - /// `rmdir` — delete an empty directory. - pub async fn rmdir(&self, base: &Arc, name: &str) -> FsResult<()> { - let ino = self.tree.lookup(base, name).await?; - if !ino.is_dir() { - return Err(FsError::NotDirectory); - } - // Emptiness check: list with max_keys=2 (we'll see the dir's own - // marker if present, plus at most one child). - let dir_key = self.tree.s3_dir_key(&ino); - let listing = self - .backend - .list_blobs(crate::backend::ListBlobsInput { - prefix: &dir_key, - delimiter: None, - max_keys: Some(2), - ..Default::default() - }) - .await?; - // Filter out the dir's own marker (key == dir_key) — that's not "content". - let real_children = listing.items.iter().filter(|i| i.key != dir_key).count(); - if real_children > 0 { - return Err(FsError::NotEmpty); - } - // Delete the marker if it exists. - if listing.items.iter().any(|i| i.key == dir_key) { - self.backend.delete_blob(&dir_key).await?; - } - self.tree.detach(&ino); - Ok(()) + /// [`Fs::lookup_at_no_follow`], confined to `scope`. + pub async fn lookup_within_no_follow( + &self, + scope: &Arc, + base: &Arc, + path: &str, + ) -> FsResult> { + self.resolve_within(scope.objid(), base, path, false).await } - /// `rename-at` — async rename. - /// - /// Returns as soon as the inode tree has been rewired and the - /// background `CopyObject` + `DeleteObject` work has been enqueued. - /// During the in-flight window, reads/writes against the new path - /// resolve to the OLD S3 key (via [`InodeTree::current_s3_key`]) so - /// the object the caller can actually touch never disappears. + /// [`Fs::parent_of`], confined to `scope`. /// - /// Errors from the worker surface on the next `Fs::sync` of the - /// renamed inode (or on a subsequent `Fs::rename` of the same inode). - /// - /// **Pre-flush.** Any open write handle for the source inode is - /// `sync`'d synchronously before the worker is enqueued — this avoids - /// a torn read where the worker's `CopyObject` runs while UploadParts - /// for an in-flight MPU on the same key are still pending. - /// - /// **Not POSIX-atomic.** Both source and destination keys briefly - /// coexist between copy and delete; a crash in that window leaves both. - pub async fn rename( + /// The scope is its own parent, so a descriptor at the top of a tenant's + /// subtree cannot be walked upwards out of it — `..` is not the only way + /// to ask for a parent, and this is the other one. + pub async fn parent_within( &self, - old_base: &Arc, - old_name: &str, - new_base: &Arc, - new_name: &str, - ) -> FsResult<()> { - let old_ino = self.tree.lookup(old_base, old_name).await?; - path::validate_segment(new_name)?; - if !new_base.is_dir() { - return Err(FsError::NotDirectory); - } - - // Surface a prior worker error or reject duplicate concurrent - // rename of the same inode. - if let Some(s) = old_ino.rename_state() { - if let Some(e) = s.error.read().clone() { - *old_ino.rename_state.write() = None; // consumed - return Err(e); - } - return Err(FsError::WouldBlock); - } - - // Pre-flush any open writable handle for the source so the worker - // copies a fully-committed S3 object. - let handles_to_flush: Vec> = self - .handles - .read() - .values() - .filter(|h| h.inode.id == old_ino.id && h.flags.write) - .cloned() - .collect(); - for h in handles_to_flush { - self.sync(&h).await?; + scope: &Arc, + ino: &Arc, + ) -> FsResult> { + if ino.objid() == scope.objid() { + return Ok(scope.clone()); } + self.parent_of(ino).await + } - if old_ino.is_dir() { - return self.rename_dir_async(&old_ino, new_base, new_name).await; - } + // -- metadata ----------------------------------------------------------- - // ----- file / symlink ----- - let src_key = self.tree.s3_key(&old_ino); - let new_parent_key = self.tree.s3_key(new_base); - let dst_key = if new_parent_key.is_empty() { - new_name.to_string() - } else { - format!("{new_parent_key}/{new_name}") - }; - if src_key == dst_key { - return Ok(()); - } + pub async fn stat(&self, ino: &Arc) -> FsResult { + Attrs::from_dnode(&self.dnode(ino.objid()).await?) + } - let attrs = old_ino.attrs.read().clone(); - let new_ino = match &*old_ino.kind.read() { - InodeKind::RegularFile => Inode::new_file( - self.tree.alloc_id(), - new_name, - Arc::downgrade(new_base), - attrs, - ), - InodeKind::Symlink { target } => Inode::new_symlink( - self.tree.alloc_id(), - new_name, - Arc::downgrade(new_base), - target.clone(), - attrs, - ), - InodeKind::Directory { .. } => unreachable!("handled above"), - }; - // Park rename state on the new inode BEFORE attaching so any - // concurrent lookup that races us sees the redirect. - *new_ino.rename_state.write() = - Some(crate::inode::attrs::RenameState::new(src_key.clone())); - - self.tree.detach(&old_ino); - self.pool.forget_inode(old_ino.id); - let attached = self.tree.attach(new_base, new_ino.clone()); - - // Enqueue the background copy+delete. If the queue is closed - // (Fs being torn down), unwind: clear state, fall through to a - // synchronous copy+delete to maintain a consistent S3 view. - if let Err(_e) = self.rename_queue.enqueue(crate::rename::RenameJob::File { - backend: self.backend.clone(), - inode: attached.clone(), - old_key: src_key.clone(), - new_key: dst_key.clone(), - }) { - *attached.rename_state.write() = None; - self.backend - .copy_blob(CopyBlobInput { - source_key: src_key.clone(), - destination_key: dst_key, - replace_metadata: None, - replace_content_type: None, - }) - .await?; - self.backend.delete_blob(&src_key).await?; - } - Ok(()) + /// The directory containing `ino`. The root is its own parent. + pub async fn parent_of(&self, ino: &Arc) -> FsResult> { + let d = self.dnode(ino.objid()).await?; + Ok(Inode::new(d.parent_objid, 0)) } - /// Recursive directory rename — async. Lists every key under the - /// source's `prefix/`, copies each to the corresponding destination - /// key, deletes the source. Cross-bucket rename and rename-into-self - /// are rejected synchronously before enqueueing the worker. - async fn rename_dir_async( + /// Resolve a path to the object itself, for callers that need identity + /// rather than attributes. + pub async fn resolve_for_hash( &self, - old_ino: &Arc, - new_base: &Arc, - new_name: &str, - ) -> FsResult<()> { - // Reject if destination exists as a non-empty dir or as a file. - if let Ok(existing) = self.tree.lookup(new_base, new_name).await { - if existing.is_dir() { - let dst_dir_key = self.tree.s3_dir_key(&existing); - let probe = self - .backend - .list_blobs(crate::backend::ListBlobsInput { - prefix: &dst_dir_key, - delimiter: None, - max_keys: Some(2), - ..Default::default() - }) - .await?; - let real = probe.items.iter().filter(|i| i.key != dst_dir_key).count(); - if real > 0 { - return Err(FsError::NotEmpty); - } - if probe.items.iter().any(|i| i.key == dst_dir_key) { - self.backend.delete_blob(&dst_dir_key).await?; - } - self.tree.detach(&existing); - } else { - return Err(FsError::NotDirectory); - } - } - - let old_prefix = format!("{}/", self.tree.s3_key(old_ino)); - let new_parent_key = self.tree.s3_key(new_base); - let new_prefix = if new_parent_key.is_empty() { - format!("{new_name}/") - } else { - format!("{new_parent_key}/{new_name}/") - }; - if old_prefix == new_prefix { - return Ok(()); - } - if new_prefix.starts_with(&old_prefix) { - return Err(FsError::Invalid("cannot rename a directory into itself")); - } + base: &Arc, + path: &str, + follow: bool, + ) -> FsResult> { + self.resolve(base, path, follow).await + } - let attrs = old_ino.attrs.read().clone(); - let explicit = old_ino.dir_explicit_marker().unwrap_or(false); - let new_ino = Inode::new_dir( - self.tree.alloc_id(), - new_name, - Some(Arc::downgrade(new_base)), - explicit, - attrs, - ); - // The "old key" for a directory rename is the prefix; current_s3_key - // doesn't apply to listings (which use new_prefix immediately) but - // the bookkeeping marker keeps duplicate-rename detection working. - *new_ino.rename_state.write() = - Some(crate::inode::attrs::RenameState::new(old_prefix.clone())); - - self.tree.detach(old_ino); - let attached = self.tree.attach(new_base, new_ino); - - if let Err(_e) = self.rename_queue.enqueue(crate::rename::RenameJob::Dir { - backend: self.backend.clone(), - inode: attached.clone(), - old_prefix: old_prefix.clone(), - new_prefix: new_prefix.clone(), - }) { - // Queue closed: unwind to synchronous directory rename. - *attached.rename_state.write() = None; - self.copy_dir_sync(&old_prefix, &new_prefix).await?; - } - Ok(()) + pub async fn stat_at(&self, base: &Arc, path: &str, follow: bool) -> FsResult { + let ino = self.resolve(base, path, follow).await?; + self.stat(&ino).await } - /// Synchronous fallback used when the rename queue is unavailable - /// (Fs being torn down). Same algorithm as the worker. - async fn copy_dir_sync(&self, old_prefix: &str, new_prefix: &str) -> FsResult<()> { - let mut continuation: Option = None; - loop { - let listing = self - .backend - .list_blobs(crate::backend::ListBlobsInput { - prefix: old_prefix, - delimiter: None, - continuation_token: continuation.as_deref(), - max_keys: None, - ..Default::default() - }) - .await?; - for item in &listing.items { - let suffix = match item.key.strip_prefix(old_prefix) { - Some(s) => s, - None => continue, - }; - let dst_key = format!("{new_prefix}{suffix}"); - self.backend - .copy_blob(CopyBlobInput { - source_key: item.key.clone(), - destination_key: dst_key, - replace_metadata: None, - replace_content_type: None, - }) - .await?; - self.backend.delete_blob(&item.key).await?; - } - if !listing.is_truncated { - break; - } - match listing.next_continuation_token { - Some(t) => continuation = Some(t), - None => break, - } + /// Directory contents, in name order. + pub async fn read_dir(&self, dir: &Arc) -> FsResult> { + let blocks = self.blocks(); + let d = self.dnode(dir.objid()).await?; + if d.kind != DnodeKind::Dir { + return Err(FsError::NotDirectory); } - Ok(()) + let mut txn = DirTxn::load(&blocks, &d).await?; + txn.list() + .await? + .into_iter() + .map(|e| { + Ok(DirEntry { + name: e.name, + kind: InodeKind::from_dnode(e.kind)?, + objid: e.objid, + }) + }) + .collect() } - /// Wait for any in-flight async rename on `inode` to complete. Returns - /// the worker's error if the rename failed (and clears it, so the - /// next call doesn't re-fire). Cheap no-op if no rename is in flight. - pub async fn wait_for_rename(&self, inode: &Arc) -> FsResult<()> { - loop { - let s = match inode.rename_state() { - Some(s) => s, - None => return Ok(()), - }; - // If the worker already finished with an error, surface it. - if let Some(e) = s.error.read().clone() { - *inode.rename_state.write() = None; - return Err(e); - } - // Otherwise wait for the next notify, then re-check. - s.done.notified().await; + // -- namespace mutation -------------------------------------------------- + + pub async fn mkdir(&self, base: &Arc, name: &str) -> FsResult> { + validate_segment(name)?; + let mut txn = self.store.begin().await?; + let blocks = txn.blocks().clone(); + let objset = txn.objset().clone(); + + let parent = objset.get_allocated(&blocks, base.objid()).await?; + require_dir(&parent)?; + let mut dir = DirTxn::load(&blocks, &parent).await?; + if dir.lookup(name).await?.is_some() { + return Err(FsError::AlreadyExists); } - } - // ------------------- symlinks ------------------- + let now = now_nanos(); + let objid = txn.reserve_objid()?; + let mut child = Dnode::new(objid, DnodeKind::Dir, parent.record_shift, now); + child.mode = DEFAULT_DIR_MODE; + child.parent_objid = parent.objid; + + dir.insert(Dirent { + name: name.to_string(), + objid, + kind: DnodeKind::Dir, + }) + .await?; + let mut new_parent = dir.finish(txn.writer()).await?; + touch(&mut new_parent, now); + + txn.stage(new_parent); + txn.stage(child); + txn.commit().await?; + Ok(Inode::new(objid, 0)) + } - /// `symlink-at` — create a new symlink at `name` under `base` pointing - /// at `target`. Atomic via `If-None-Match: *` on the underlying PUT; - /// returns `AlreadyExists` if the destination exists. pub async fn symlink_at( &self, base: &Arc, name: &str, target: &str, ) -> FsResult> { - if !base.is_dir() { - return Err(FsError::NotDirectory); - } - if base.is_deleted() { - return Err(FsError::NotFound); + validate_segment(name)?; + validate_symlink_target(target)?; + + let mut txn = self.store.begin().await?; + let blocks = txn.blocks().clone(); + let objset = txn.objset().clone(); + + let parent = objset.get_allocated(&blocks, base.objid()).await?; + require_dir(&parent)?; + let mut dir = DirTxn::load(&blocks, &parent).await?; + if dir.lookup(name).await?.is_some() { + return Err(FsError::AlreadyExists); } - path::validate_segment(name)?; - path::validate_symlink_target(target)?; - let parent_key = self.tree.s3_key(base); - let key = if parent_key.is_empty() { - name.to_string() - } else { - format!("{parent_key}/{name}") - }; + let now = now_nanos(); + let objid = txn.reserve_objid()?; + let mut link = Dnode::new(objid, DnodeKind::Symlink, parent.record_shift, now); + link.mode = DEFAULT_SYMLINK_MODE; + link.parent_objid = parent.objid; + link.size = target.len() as u64; - let mut metadata = HashMap::new(); - metadata.insert( - SYMLINK_METADATA_KEY.to_string(), - SYMLINK_METADATA_VALUE.to_string(), - ); - - let meta = self - .backend - .put_blob_if_not_exists(PutBlobInput { - key, - body: Bytes::copy_from_slice(target.as_bytes()), - metadata, - content_type: Some(SYMLINK_CONTENT_TYPE.to_string()), - }) - .await?; + let bytes = target.as_bytes(); + if bytes.len() <= INLINE_CAP { + // Almost every target fits here, so readlink costs no block read. + link.inline = bytes.to_vec(); + } else { + let record = parent.record_size(); + let dirty: BTreeMap> = bytes + .chunks(record) + .enumerate() + .map(|(i, c)| (i as u64, c.to_vec())) + .collect(); + link = commit_object(&blocks, txn.writer(), &link, dirty, bytes.len() as u64).await?; + } + + dir.insert(Dirent { + name: name.to_string(), + objid, + kind: DnodeKind::Symlink, + }) + .await?; + let mut new_parent = dir.finish(txn.writer()).await?; + touch(&mut new_parent, now); - let attrs = attrs_from_meta(&meta); - let id = self.tree.alloc_id(); - let inode = Inode::new_symlink(id, name, Arc::downgrade(base), target.to_string(), attrs); - Ok(self.tree.attach(base, inode)) + txn.stage(new_parent); + txn.stage(link); + txn.commit().await?; + Ok(Inode::new(objid, 0)) } - /// `readlink-at` — return the literal target string of a symlink. - /// Errors with `Invalid` if the target is not a symlink. pub async fn readlink_at(&self, base: &Arc, name: &str) -> FsResult { - let ino = self.tree.lookup(base, name).await?; - match ino.symlink_target() { - Some(t) => Ok(t), - None => Err(FsError::Invalid("readlink_at: not a symlink")), + let ino = self.resolve(base, name, false).await?; + let d = self.dnode(ino.objid()).await?; + if d.kind != DnodeKind::Symlink { + return Err(FsError::Invalid("not a symlink")); } + read_symlink(&self.blocks(), &d).await } - // ------------------- set_size / set_times ------------------- - - /// `set-size` — truncate or grow a file to exactly `new_size` bytes. + /// Add a second name for an existing object. /// - /// **Shrink:** any pending writes are first synced to S3 (so we don't - /// lose them by truncating mid-buffer); then we issue a ranged `GET` - /// for `[0, new_size)` and re-`PUT` the result. Simple and correct for - /// any size, but downloads + re-uploads the surviving prefix. - /// Acceptable for typical truncate workloads (`> file`, log rotation, - /// SQLite VACUUM); large in-place shrinks would benefit from an MPU - /// rewrite path which we can layer on later. + /// Only files and symlinks: a hard link to a directory would make the + /// namespace a graph rather than a tree, and the parent link in the dnode + /// has room for exactly one answer. /// - /// **Grow:** zero-fills the gap by issuing a normal `pwrite` of zeros at - /// the previous EOF. Capped at 100 MiB to avoid blowing up memory; for - /// larger growths use direct `pwrite` of your real bytes. - pub async fn set_size(&self, handle: &FileHandle, new_size: u64) -> FsResult<()> { - if !handle.flags.write { - return Err(FsError::AccessDenied); + /// This was permanently unsupported under the previous engine, which had + /// no way to express shared identity across two S3 keys. Here a directory + /// entry is already just an object id, so a link is one more entry and an + /// increment of `nlink`. + pub async fn link_at( + &self, + old_base: &Arc, + old_path: &str, + new_base: &Arc, + new_name: &str, + follow: bool, + ) -> FsResult<()> { + self.link_within(&self.root(), old_base, old_path, new_base, new_name, follow) + .await + } + + /// [`Fs::link_at`], confined to `scope`. + /// + /// The *source* is the one that matters here: a hard link is a second name + /// for an existing inode, so linking to something outside the scope would + /// pull it inside permanently — an escape that survives the request that + /// made it. + #[allow(clippy::too_many_arguments)] + pub async fn link_within( + &self, + scope: &Arc, + old_base: &Arc, + old_path: &str, + new_base: &Arc, + new_name: &str, + follow: bool, + ) -> FsResult<()> { + validate_segment(new_name)?; + let target = self + .resolve_within(scope.objid(), old_base, old_path, follow) + .await?; + + let mut txn = self.store.begin().await?; + let blocks = txn.blocks().clone(); + let objset = txn.objset().clone(); + + let mut victim = objset.get_allocated(&blocks, target.objid()).await?; + if victim.kind == DnodeKind::Dir { + return Err(FsError::NotPermitted); } - if handle.inode.is_deleted() { - return Err(FsError::NotFound); + + let parent = objset.get_allocated(&blocks, new_base.objid()).await?; + require_dir(&parent)?; + let mut dir = DirTxn::load(&blocks, &parent).await?; + if dir.lookup(new_name).await?.is_some() { + return Err(FsError::AlreadyExists); } - let current = *handle.size.read(); - if new_size == current { - return Ok(()); + + let now = now_nanos(); + dir.insert(Dirent { + name: new_name.to_string(), + objid: victim.objid, + kind: victim.kind, + }) + .await?; + victim.nlink = victim.nlink.saturating_add(1); + victim.ctime_nanos = now; + + let mut new_parent = dir.finish(txn.writer()).await?; + touch(&mut new_parent, now); + txn.stage(new_parent); + txn.stage(victim); + txn.commit().await.map(|_| ()) + } + + pub async fn unlink(&self, base: &Arc, name: &str) -> FsResult<()> { + self.remove_entry(base, name, false).await + } + + pub async fn rmdir(&self, base: &Arc, name: &str) -> FsResult<()> { + self.remove_entry(base, name, true).await + } + + async fn remove_entry(&self, base: &Arc, name: &str, want_dir: bool) -> FsResult<()> { + validate_segment(name)?; + let mut txn = self.store.begin().await?; + let blocks = txn.blocks().clone(); + let objset = txn.objset().clone(); + + let parent = objset.get_allocated(&blocks, base.objid()).await?; + require_dir(&parent)?; + let mut dir = DirTxn::load(&blocks, &parent).await?; + let entry = dir.lookup(name).await?.ok_or(FsError::NotFound)?; + + let is_dir = entry.kind == DnodeKind::Dir; + match (want_dir, is_dir) { + (true, false) => return Err(FsError::NotDirectory), + (false, true) => return Err(FsError::IsDirectory), + _ => {} } - if new_size < current { - // Shrink: flush any pending writes first. - self.sync(handle).await?; - let key = self.tree.current_s3_key(&handle.inode); - let body = if new_size == 0 { - Bytes::new() + let mut child = objset.get_allocated(&blocks, entry.objid).await?; + if is_dir { + let child_dir = DirTxn::load(&blocks, &child).await?; + if !child_dir.is_empty() { + return Err(FsError::NotEmpty); + } + } + + dir.remove(name).await?; + let now = now_nanos(); + let mut new_parent = dir.finish(txn.writer()).await?; + touch(&mut new_parent, now); + + child.nlink = child.nlink.saturating_sub(1); + child.ctime_nanos = now; + if child.nlink == 0 { + // Last name gone. Keep the object alive if a handle still holds + // it — POSIX promises an unlinked file stays readable until the + // last descriptor closes. With no entries left, the only way to + // reach it is a handle that already exists, so no new opener can + // find it in the meantime. + let mut open = self.open.lock(); + if open.is_open(child.objid) { + open.mark_orphan(child.objid); } else { - self.backend.get_blob(&key, Some(0..new_size)).await?.body - }; - let meta = self - .backend - .put_blob(PutBlobInput { - key, - body, - metadata: HashMap::new(), - content_type: None, - }) - .await?; - *handle.inode.attrs.write() = attrs_from_meta(&meta); - *handle.size.write() = new_size; - self.pool.forget_inode(handle.inode.id); - Ok(()) - } else { - // Grow: zero-fill via the buffer pool. Cap to avoid OOM. - let extra = new_size - current; - const MAX_GROW_BYTES: u64 = 100 * 1024 * 1024; - if extra > MAX_GROW_BYTES { - return Err(FsError::Invalid( - "set_size grow exceeds 100 MiB cap; use pwrite of real bytes", - )); + // Its blocks become unreachable copy-on-write garbage, which + // lifecycle policy on the data bucket reclaims. Nothing is + // deleted here. + child = Dnode::free(child.objid, child.record_shift); } - let zeros = vec![0u8; extra as usize]; - self.pwrite(handle, current, &zeros).await?; - Ok(()) } + + txn.stage(new_parent); + txn.stage(child); + txn.commit().await.map(|_| ()).map(|_| ()) } - /// `set-times-at` — persist atime/mtime as user metadata via a - /// `CopyObject` self-copy with `MetadataDirective=REPLACE`. The values - /// land in `x-amz-meta-s3wasifs-{atime,mtime}` as RFC-3339 nanos. - /// - /// `None` means "don't change"; `Some(t)` sets the field. The cached - /// inode attrs are updated so subsequent `stat` calls see the new - /// times within the same process — across mounts, the times are - /// readable on next `lookup` (which fetches metadata via `HeadObject`). - pub async fn set_times_at( + /// Move an entry. Atomic: one directory-entry move inside one commit. + pub async fn rename( &self, - base: &Arc, - path: &str, - follow_symlinks: bool, - atime: Option, - mtime: Option, + from_base: &Arc, + from_name: &str, + to_base: &Arc, + to_name: &str, ) -> FsResult<()> { - let ino = if follow_symlinks { - self.tree.lookup_at(base, path).await? + validate_segment(from_name)?; + validate_segment(to_name)?; + let (src_id, dst_id) = (from_base.objid(), to_base.objid()); + if src_id == dst_id && from_name == to_name { + return Ok(()); + } + + let mut txn = self.store.begin().await?; + let blocks = txn.blocks().clone(); + let objset = txn.objset().clone(); + let now = now_nanos(); + + let src_dnode = objset.get_allocated(&blocks, src_id).await?; + require_dir(&src_dnode)?; + let mut src_dir = DirTxn::load(&blocks, &src_dnode).await?; + let entry = src_dir.lookup(from_name).await?.ok_or(FsError::NotFound)?; + + // A directory may not move into itself: the subtree would keep its + // parent link into a directory that now lives inside it, and nothing + // from the root could reach any of it again. Walk up from the + // destination; the root is its own parent, which ends the walk. + if entry.kind == DnodeKind::Dir { + let mut cursor = dst_id; + loop { + if cursor == entry.objid { + return Err(FsError::Invalid("cannot rename a directory into itself")); + } + let parent = objset.get_allocated(&blocks, cursor).await?.parent_objid; + if parent == cursor { + break; + } + cursor = parent; + } + } + + // Same directory: one transaction over one structure, or the two would + // each rebuild from the same base and the second would erase the first. + let mut dst_dir = if src_id == dst_id { + None } else { - self.tree.lookup_at_no_follow(base, path).await? + let d = objset.get_allocated(&blocks, dst_id).await?; + require_dir(&d)?; + Some(DirTxn::load(&blocks, &d).await?) }; - self.set_times_inode(&ino, atime, mtime).await - } - /// Like `set_times_at` but on an already-resolved file handle. - pub async fn set_times( - &self, - handle: &FileHandle, - atime: Option, - mtime: Option, - ) -> FsResult<()> { - self.set_times_inode(&handle.inode, atime, mtime).await - } + let existing = match &mut dst_dir { + Some(d) => d.lookup(to_name).await?, + None => src_dir.lookup(to_name).await?, + }; + if let Some(victim) = existing { + if victim.objid == entry.objid { + return Ok(()); // already linked here + } + let mut victim_dnode = objset.get_allocated(&blocks, victim.objid).await?; + match (entry.kind == DnodeKind::Dir, victim.kind == DnodeKind::Dir) { + (true, false) => return Err(FsError::NotDirectory), + (false, true) => return Err(FsError::IsDirectory), + (true, true) => { + if !DirTxn::load(&blocks, &victim_dnode).await?.is_empty() { + return Err(FsError::NotEmpty); + } + } + (false, false) => {} + } + match &mut dst_dir { + Some(d) => d.remove(to_name).await?, + None => src_dir.remove(to_name).await?, + }; + victim_dnode.nlink = victim_dnode.nlink.saturating_sub(1); + victim_dnode.ctime_nanos = now; + if victim_dnode.nlink == 0 { + // Same contract as unlink: a replaced file that is still open + // survives until its last handle closes. + let mut open = self.open.lock(); + if open.is_open(victim_dnode.objid) { + open.mark_orphan(victim_dnode.objid); + } else { + victim_dnode = Dnode::free(victim_dnode.objid, victim_dnode.record_shift); + } + } + txn.stage(victim_dnode); + } - async fn set_times_inode( - &self, - ino: &Arc, - atime: Option, - mtime: Option, - ) -> FsResult<()> { - if atime.is_none() && mtime.is_none() { - return Ok(()); + src_dir.remove(from_name).await?; + let moved = Dirent { + name: to_name.to_string(), + objid: entry.objid, + kind: entry.kind, + }; + match &mut dst_dir { + Some(d) => d.insert(moved).await?, + None => src_dir.insert(moved).await?, } - let key = self.tree.current_s3_key(ino); - // Read current head to preserve metadata fields we don't touch. - let head = self.backend.head_blob(&key).await?; - let mut metadata = head.metadata.clone(); - let mut new_attrs = ino.attrs.read().clone(); - if let Some(t) = atime { - metadata.insert(METADATA_ATIME_KEY.to_string(), systemtime_to_meta(t)); + + // A directory carries its parent in its dnode, so moving one across + // directories has to update that link or `..` would still point home. + if entry.kind == DnodeKind::Dir && src_id != dst_id { + let mut child = objset.get_allocated(&blocks, entry.objid).await?; + child.parent_objid = dst_id; + child.ctime_nanos = now; + txn.stage(child); } - if let Some(t) = mtime { - metadata.insert(METADATA_MTIME_KEY.to_string(), systemtime_to_meta(t)); - new_attrs.last_modified = t; + + let mut new_src = src_dir.finish(txn.writer()).await?; + touch(&mut new_src, now); + txn.stage(new_src); + if let Some(d) = dst_dir { + let mut new_dst = d.finish(txn.writer()).await?; + touch(&mut new_dst, now); + txn.stage(new_dst); } - self.backend - .copy_blob(CopyBlobInput { - source_key: key.clone(), - destination_key: key, - replace_metadata: Some(metadata), - replace_content_type: head.content_type, - }) - .await?; - *ino.attrs.write() = new_attrs; - Ok(()) + txn.commit().await.map(|_| ()) } - // ------------------- open / close ------------------- + // -- files --------------------------------------------------------------- - /// `open-at` — resolve `path` from `base`, honour open flags, return a - /// fresh `FileHandle`. pub async fn open_at( &self, base: &Arc, path: &str, flags: OpenFlags, ) -> FsResult> { - // Validate flag combo: at least one of read/write must be set. - if !flags.read && !flags.write { - return Err(FsError::Invalid("open with no read or write")); - } - // O_EXCL implies create. - if flags.exclusive && !flags.create { - return Err(FsError::Invalid("O_EXCL without O_CREAT")); - } + self.open_within(&self.root(), base, path, flags).await + } - // First, try lookup. If found, handle truncate/exclusive; if absent - // and create requested, create. - let ino = match self.tree.lookup_at(base, path).await { - Ok(i) => { - if flags.exclusive { + /// [`Fs::open_at`], confined to `scope`. + /// + /// Both halves have to be confined, not just the lookup: a path that does + /// not resolve may still be *created*, and creating + /// `../someone-else/file` would be an escape that writes rather than + /// reads. + pub async fn open_within( + &self, + scope: &Arc, + base: &Arc, + path: &str, + flags: OpenFlags, + ) -> FsResult> { + if !flags.read && !flags.write { + return Err(FsError::Invalid("open with no read or write")); + } + + let existing = match self.resolve_within(scope.objid(), base, path, true).await { + Ok(ino) => Some(ino), + Err(FsError::NotFound) if flags.create => None, + Err(e) => return Err(e), + }; + + let inode = match existing { + Some(ino) => { + if flags.exclusive { return Err(FsError::AlreadyExists); } - if i.is_dir() && (flags.write || flags.truncate) { - return Err(FsError::IsDirectory); - } - i + ino } - Err(FsError::NotFound) if flags.create => self.create_file_at(base, path).await?, - Err(e) => return Err(e), + None => self.create_file(scope.objid(), base, path).await?, }; - // O_TRUNC: synchronously zero the file via PutObject of empty bytes. - if flags.truncate && ino.is_regular_file() { - let key = self.tree.s3_key(&ino); - let meta = self - .backend - .put_blob(PutBlobInput { - key, - body: Bytes::new(), - metadata: HashMap::new(), - content_type: None, - }) - .await?; - let new_attrs = attrs_from_meta(&meta); - *ino.attrs.write() = new_attrs; - self.pool.forget_inode(ino.id); + let d = self.dnode(inode.objid()).await?; + if d.kind == DnodeKind::Dir && flags.write { + return Err(FsError::IsDirectory); } - let id = self.alloc_handle_id(); - let h = Arc::new(FileHandle::new(id, ino, flags)); - self.handles.write().insert(id, h.clone()); - Ok(h) + let handle = self.register_handle(inode, flags, &d); + if flags.truncate && flags.write { + self.set_size(&handle, 0).await?; + } + Ok(handle) } - /// Convenience: `open_at(root, path, flags)`. pub async fn open(&self, path: &str, flags: OpenFlags) -> FsResult> { - let root = self.root(); - self.open_at(&root, path, flags).await + self.open_at(&self.root(), path, flags).await } - /// Drop a handle. Does NOT sync; pending dirty writes are abandoned and - /// any in-flight MPU is aborted to avoid leaking partial uploads. - pub async fn close(&self, handle: &Arc) -> FsResult<()> { - // If a sync was never called, abort any in-flight MPU. - let mut mpu = handle.mpu.lock().await; - if let Some(state) = mpu.as_mut() { - mpu::abort(state, &*self.backend).await?; + /// Create an empty regular file at `path`, which must not exist. + async fn create_file(&self, scope: u64, base: &Arc, path: &str) -> FsResult> { + let (parent_path, name) = split_last(path)?; + let parent = if parent_path.is_empty() { + base.clone() + } else { + self.resolve_within(scope, base, parent_path, true).await? + }; + validate_segment(name)?; + + let mut txn = self.store.begin().await?; + let blocks = txn.blocks().clone(); + let objset = txn.objset().clone(); + + let parent_dnode = objset.get_allocated(&blocks, parent.objid()).await?; + require_dir(&parent_dnode)?; + let mut dir = DirTxn::load(&blocks, &parent_dnode).await?; + // Another commit may have created it between our lookup and this + // transaction. The whole create is inside the transaction, so losing + // that race is visible rather than silently overwriting. + if dir.lookup(name).await?.is_some() { + return Err(FsError::AlreadyExists); } - drop(mpu); - self.handles.write().remove(&handle.id); - Ok(()) - } - /// Internal: create a new empty file at `path` under `base`. Used by - /// the `O_CREAT` path of `open_at`. Atomic via `put_blob_if_not_exists` - /// when `O_EXCL` is also set; otherwise falls back to plain `put_blob`. - async fn create_file_at(&self, base: &Arc, path: &str) -> FsResult> { - // Resolve parent + final segment. We re-walk one shy of the leaf so - // we have the final basename to attach in the inode tree. - let (parent_dir, basename) = self.resolve_parent_and_basename(base, path).await?; - path::validate_segment(&basename)?; + let now = now_nanos(); + let objid = txn.reserve_objid()?; + let mut file = Dnode::new(objid, DnodeKind::File, parent_dnode.record_shift, now); + file.mode = DEFAULT_FILE_MODE; + file.parent_objid = parent_dnode.objid; - let parent_key = self.tree.s3_key(&parent_dir); - let key = if parent_key.is_empty() { - basename.clone() - } else { - format!("{parent_key}/{basename}") - }; + dir.insert(Dirent { + name: name.to_string(), + objid, + kind: DnodeKind::File, + }) + .await?; + let mut new_parent = dir.finish(txn.writer()).await?; + touch(&mut new_parent, now); - // For plain create (without O_EXCL) we just PUT empty. The exclusive - // case is handled by the caller via lookup-then-error before reaching - // here — but to be race-safe against another writer, also use - // `put_blob_if_not_exists` if available so we don't silently - // overwrite a sibling that appeared between our lookup and our PUT. - let put = PutBlobInput { - key, - body: Bytes::new(), - metadata: HashMap::new(), - content_type: None, - }; - let meta = if self.backend.capabilities().conditional_put { - self.backend.put_blob_if_not_exists(put).await? - } else { - self.backend.put_blob(put).await? - }; + txn.stage(new_parent); + txn.stage(file); + txn.commit().await?; + Ok(Inode::new(objid, 0)) + } - let attrs = attrs_from_meta(&meta); - let id = self.tree.alloc_id(); - let inode = Inode::new_file(id, &basename, Arc::downgrade(&parent_dir), attrs); - Ok(self.tree.attach(&parent_dir, inode)) + fn register_handle(&self, inode: Arc, flags: OpenFlags, d: &Dnode) -> Arc { + let raw = self.next_handle.fetch_add(1, Ordering::Relaxed); + let id = HandleId(NonZeroU64::new(raw).expect("handle ids start at 1")); + let handle = Arc::new(FileHandle { + id, + inode, + flags, + state: Mutex::new(HandleState { + size: d.size, + dirty: BTreeMap::new(), + meta_dirty: false, + atime_nanos: d.atime_nanos, + mtime_nanos: d.mtime_nanos, + }), + }); + self.handles.lock().insert(raw, handle.clone()); + self.open.lock().opened(handle.inode.objid()); + handle } - /// Resolve `path` from `base` to `(parent_dir_inode, basename)`. - async fn resolve_parent_and_basename( - &self, - base: &Arc, - path: &str, - ) -> FsResult<(Arc, String)> { - // Split off the trailing component. - // We want POSIX-style: "a/b/c.txt" → parent="a/b", name="c.txt". - // For just "name", parent = base, name = "name". - if path.starts_with('/') { - return Err(FsError::NotPermitted); - } - // Strip a possible single trailing slash (treated as "is-a-directory" hint). - let trimmed = path.trim_end_matches('/'); - let (parent_path, name) = match trimmed.rfind('/') { - Some(i) => (&trimmed[..i], &trimmed[i + 1..]), - None => ("", trimmed), - }; - if name.is_empty() { - return Err(FsError::Invalid("empty basename")); - } - let parent_dir = if parent_path.is_empty() { - base.clone() + pub async fn close(&self, handle: &Arc) -> FsResult<()> { + let result = if handle.flags.write { + self.sync(handle).await } else { - self.tree.lookup_at(base, parent_path).await? + Ok(()) }; - if !parent_dir.is_dir() { - return Err(FsError::NotDirectory); + self.abandon(handle).await?; + result + } + + /// Close without flushing: whatever the handle has not synced is thrown + /// away, exactly as on a crash. + /// + /// For a handle whose owner went away without closing it. Its buffer is a + /// write nobody is waiting on, and flushing it whenever the release gets + /// round to it would land it on top of anything committed in the + /// meantime — an update undone by one that came before it. + pub async fn abandon(&self, handle: &Arc) -> FsResult<()> { + let objid = handle.inode.objid(); + self.handles.lock().remove(&handle.id.get()); + + // If this was the last handle on an object whose last name is already + // gone, now is when it actually goes away. + let reap = self.open.lock().closed(objid); + if reap { + self.free_orphan(objid).await?; } - Ok((parent_dir, name.to_string())) + Ok(()) } - // ------------------- pread / pwrite / sync ------------------- + /// Free an object that lost its last name while it was open. + async fn free_orphan(&self, objid: u64) -> FsResult<()> { + let mut txn = self.store.begin().await?; + let blocks = txn.blocks().clone(); + let objset = txn.objset().clone(); + let d = match objset.get_allocated(&blocks, objid).await { + Ok(d) => d, + // Already gone: another path freed it, or the mount was + // reformatted underneath us. Nothing to do. + Err(FsError::NotFound) => return Ok(()), + Err(e) => return Err(e), + }; + if d.nlink > 0 { + // It was linked again between the unlink and this close, so it is + // reachable by name once more and must not be freed. + return Ok(()); + } + txn.stage(Dnode::free(objid, d.record_shift)); + txn.commit().await.map(|_| ()) + } - /// Read `len` bytes from the file at `offset`. Returns up to `len` bytes; - /// short reads at EOF return fewer. pub async fn pread(&self, handle: &FileHandle, offset: u64, len: usize) -> FsResult { if !handle.flags.read { - return Err(FsError::AccessDenied); - } - if handle.inode.is_deleted() { - return Err(FsError::NotFound); + return Err(FsError::BadDescriptor); } - let size = *handle.size.read(); - if offset >= size || len == 0 { + let st = handle.state.lock().await; + if offset >= st.size || len == 0 { return Ok(Bytes::new()); } - let read_end = (offset + len as u64).min(size); - let mut out = BytesMut::with_capacity((read_end - offset) as usize); - - let mut cur = offset; - while cur < read_end { - let loc = self - .config - .part_schedule - .locate(cur) - .ok_or(FsError::FileTooLarge)?; - let part_arc = self.get_or_fetch_part(handle, loc.part_index).await?; - let part = part_arc.read(); - let into_part = cur - loc.part_start; - let to_read = (read_end - cur).min(loc.part_size - into_part); - let chunk = part.read(into_part, to_read); - out.extend_from_slice(&chunk); - // If the part returned fewer bytes than requested, we hit its - // valid_len early — clamp and break. - if (chunk.len() as u64) < to_read { - break; - } - cur += to_read; + let end = offset.saturating_add(len as u64).min(st.size); + + // A view of the object at the size this handle sees, so records past + // the committed end read as holes rather than as past-EOF. + let mut view = self.dnode(handle.inode.objid()).await?; + view.size = st.size; + let record = view.record_size() as u64; + let blocks = self.blocks(); + + let mut out = Vec::with_capacity((end - offset) as usize); + let mut pos = offset; + while pos < end { + let index = pos / record; + let within = (pos % record) as usize; + let take = ((record - within as u64).min(end - pos)) as usize; + + let block = match st.dirty.get(&index) { + Some(buf) => Bytes::from(buf.clone()), + None => read_data_block(&blocks, &view, index).await?, + }; + let slice = block.get(within..).unwrap_or(&[]); + let n = take.min(slice.len()); + out.extend_from_slice(&slice[..n]); + // A short record means a hole at the tail of the object; the rest + // of the requested span is zeros. + out.resize(out.len() + (take - n), 0); + pos += take as u64; } - Ok(out.freeze()) + Ok(Bytes::from(out)) } - /// Write `data` at `offset`. Returns the number of bytes written - /// (always `data.len()` on success). Marks affected parts dirty and - /// updates the handle's logical file size if the write extends past EOF. - /// Does NOT call S3 — `sync` does the actual upload. pub async fn pwrite(&self, handle: &FileHandle, offset: u64, data: &[u8]) -> FsResult { if !handle.flags.write { - return Err(FsError::AccessDenied); - } - if handle.inode.is_deleted() { - return Err(FsError::NotFound); + return Err(FsError::BadDescriptor); } if data.is_empty() { return Ok(0); } + let mut st = handle.state.lock().await; + let offset = if handle.flags.append { st.size } else { offset }; - let mut written = 0usize; - let total_end = offset + data.len() as u64; - let mut cur = offset; - let mut data_off = 0usize; - // Parts that became fully dirty during this call — eagerly enqueue - // after we drop all part locks so the await point is clean. - let mut eager_parts: Vec<(u32, Arc>)> = Vec::new(); - - while data_off < data.len() { - let loc = self - .config - .part_schedule - .locate(cur) - .ok_or(FsError::FileTooLarge)?; - let part_arc = self - .get_or_fetch_part_for_write(handle, loc.part_index) - .await?; - - // If this part has an in-flight eager upload, await it before - // mutating. This both bounds memory pressure and prevents a - // mark_flushing → apply_write WouldBlock race. - let inflight = part_arc.read().take_inflight(); - if let Some(rx) = inflight { - self.absorb_inflight(handle, rx).await?; - } + let mut view = self.dnode(handle.inode.objid()).await?; + view.size = st.size; + let record = view.record_size(); + let blocks = self.blocks(); - let into_part = cur - loc.part_start; - let space_in_part = loc.part_size - into_part; - let to_write = ((data.len() - data_off) as u64).min(space_in_part) as usize; - - let became_full = { - let mut part = part_arc.write(); - part.apply_write(into_part, &data[data_off..data_off + to_write])?; - // Only eagerly enqueue when the part is COMPLETELY filled - // to its tier capacity (not just "dirty covers valid_len" — - // that's true for a partial last part too, and uploading - // those wastes work the next pwrite would overwrite). - let fully_filled = part.valid_len == part.part_size && part.is_fully_dirty(); - let no_inflight = part - .upload_in_flight - .lock() - .ok() - .is_some_and(|g| g.is_none()); - fully_filled && no_inflight + let mut written = 0usize; + while written < data.len() { + let pos = offset + written as u64; + let index = pos / record as u64; + let within = (pos % record as u64) as usize; + let take = (record - within).min(data.len() - written); + + let mut buf = match st.dirty.remove(&index) { + Some(buf) => buf, + None => read_data_block(&blocks, &view, index).await?.to_vec(), }; - - if became_full { - eager_parts.push((loc.part_index, part_arc.clone())); - } - - data_off += to_write; - cur += to_write as u64; - written += to_write; - } - - // Update logical file size BEFORE submitting eager uploads — the - // worker reads `final_size` from the snapshot we pass it. - { - let mut sz = handle.size.write(); - if total_end > *sz { - *sz = total_end; - } - } - - // Eager flush: enqueue any newly-full parts. Skip when the file is - // small enough that single-PUT will win (no point starting an MPU). - let final_size = *handle.size.read(); - if !eager_parts.is_empty() && final_size > self.config.single_part_threshold { - self.enqueue_eager_uploads(handle, eager_parts).await?; + // Records are held at full width while dirty and trimmed to the + // object's real tail length at commit, so partial writes into a + // hole do not have to reason about the boundary. + buf.resize(record, 0); + buf[within..within + take].copy_from_slice(&data[written..written + take]); + st.dirty.insert(index, buf); + written += take; } + st.size = st.size.max(offset + data.len() as u64); + st.mtime_nanos = now_nanos(); Ok(written) } - /// Lazily begin the per-handle MPU and enqueue an `UploadPart` job for - /// each newly-full part, parking the reply receiver on the part so a - /// later `sync` can await it. - async fn enqueue_eager_uploads( - &self, - handle: &FileHandle, - parts: Vec<(u32, Arc>)>, - ) -> FsResult<()> { - let key = self.tree.current_s3_key(&handle.inode); - - // Begin MPU under handle.mpu lock if not already started. - let mut mpu_guard = handle.mpu.lock().await; - let (source_size, source_etag) = { - let a = handle.inode.attrs.read(); - if a.etag.is_empty() { - (None, None) - } else { - (Some(a.size), Some(a.etag.clone())) - } - }; - let upload_id = crate::flusher::ensure_mpu_begun_locked( - &self.backend, - &mut mpu_guard, - &key, - &self.config.part_schedule, - source_size, - source_etag, - ) - .await?; - // Note the source_size from MpuState — it may differ from the - // current inode attrs after concurrent activity. - let mpu_source_size = mpu_guard.as_ref().and_then(|s| s.source_size).unwrap_or(0); - drop(mpu_guard); - - let final_size = *handle.size.read(); - for (part_index, part_arc) in parts { - let rx = self.flusher.enqueue_upload( - self.backend.clone(), - key.clone(), - upload_id.clone(), - part_index, - part_arc.clone(), - self.config.part_schedule.clone(), - mpu_source_size, - final_size, - )?; - part_arc.read().park_inflight(rx); - } - Ok(()) - } - - /// Await an in-flight `UploadPart` reply, fold the result into the - /// per-handle MPU state, and surface errors to the caller. On error the - /// part's state is rolled back to Dirty so a later `sync` can retry. - async fn absorb_inflight( - &self, - handle: &FileHandle, - rx: tokio::sync::oneshot::Receiver, - ) -> FsResult<()> { - match rx.await { - Ok(Ok((part_index, etag))) => { - let mut g = handle.mpu.lock().await; - if let Some(state) = g.as_mut() { - state.record_part_uploaded(part_index, etag); - } - Ok(()) - } - Ok(Err(e)) => Err(e), - // Worker dropped without replying (shouldn't happen unless Fs - // was torn down). Treat as transient I/O. - Err(_) => Err(FsError::Io("flush worker dropped before reply".into())), - } - } - - /// Flush dirty bytes to S3 and finalise. Durable when this returns. pub async fn sync(&self, handle: &FileHandle) -> FsResult<()> { - if handle.inode.is_deleted() { - return Err(FsError::NotFound); - } - if !handle.flags.write { - // Read-only handle: nothing to sync. + let mut st = handle.state.lock().await; + if st.dirty.is_empty() && !st.meta_dirty { return Ok(()); } - - // Hold the per-inode rename lock for the whole sync so a - // concurrent rename worker can't run its CopyObject on a key that - // we're mid-commit on. The worker takes the same lock; ordering - // is "first acquired, first served". Cheap when no rename is in - // flight (uncontested mutex). - let _rename_guard = handle.inode.rename_lock.lock().await; - - // If a prior rename worker errored (and left state set), surface - // and clear so the caller sees it once. - if let Some(s) = handle.inode.rename_state() { - if let Some(e) = s.error.read().clone() { - *handle.inode.rename_state.write() = None; - return Err(e); - } - } - - // Drain any eager-upload receivers parked on parts. After this, - // every part is either Clean/Flushed (background upload done and - // recorded into MpuState) or Dirty (no in-flight, needs sync to - // upload it inline). - self.drain_inflight_uploads(handle).await?; - - let final_size = *handle.size.read(); - let key = self.tree.current_s3_key(&handle.inode); - - // Collect dirty parts. We snapshot under the pool's per-part read - // locks; the buffer pool itself is already lock-light. - let mut dirty_parts: Vec<(u32, Arc>)> = Vec::new(); - // We don't have a "list parts for inode" API on the pool; we walk - // possible part indices up to the final size's high water. - if final_size > 0 { - let high = self - .config - .part_schedule - .locate(final_size - 1) - .ok_or(FsError::FileTooLarge)? - .part_index; - for i in 0..=high { - let key = PartKey::new(handle.inode.id, i); - if let Some(part_arc) = self.pool.get(key) { - let st = part_arc.read().state.clone(); - if matches!(st, crate::buffer::PartState::Dirty) { - dirty_parts.push((i, part_arc)); - } + self.flush(handle.inode.objid(), &mut st).await + } + + async fn flush(&self, objid: u64, st: &mut HandleState) -> FsResult<()> { + let mut txn = self.store.begin().await?; + let blocks = txn.blocks().clone(); + let objset = txn.objset().clone(); + + // Re-read rather than trusting the copy taken at open: another commit + // may have advanced this object since. + let base = objset.get_allocated(&blocks, objid).await?; + let record = base.record_size(); + + // A copy, not a take: if the commit fails the buffers must still be + // here for the retry, or a sync that timed out once loses the file. + let mut dirty = st.dirty.clone(); + let nblocks = blocks_for_size(st.size, record); + dirty.retain(|index, _| *index < nblocks); + for (index, buf) in dirty.iter_mut() { + buf.resize(block_logical_len(st.size, record, *index), 0); + } + + // Shrinking has to rewrite the block the new end lands in, not merely + // record a smaller size. Blocks past the end are dropped by the + // copy-on-write rebuild, but the boundary block keeps whatever it held + // — so a later grow would read those bytes back out from under the + // truncation instead of the zeros POSIX promises. + if st.size < base.size && nblocks > 0 { + let index = nblocks - 1; + if let std::collections::btree_map::Entry::Vacant(slot) = dirty.entry(index) { + let want = block_logical_len(st.size, record, index); + let mut buf = read_data_block(&blocks, &base, index).await?.to_vec(); + if buf.len() > want { + buf.truncate(want); + slot.insert(buf); } } } - // No dirty bytes and no in-flight MPU → nothing to do. - let mpu_in_flight = handle - .mpu - .lock() - .await - .as_ref() - .is_some_and(|s| s.has_upload()); - if dirty_parts.is_empty() && !mpu_in_flight { - return Ok(()); - } - - // Decide the commit path. - // Small-file fast path: file size below threshold AND no MPU started. - if !mpu_in_flight && final_size <= self.config.single_part_threshold { - self.sync_via_single_put(handle, &key, final_size, dirty_parts) - .await?; - return Ok(()); - } - - // MPU path. - self.sync_via_mpu(handle, &key, dirty_parts).await - } + let mut updated = commit_object(&blocks, txn.writer(), &base, dirty, st.size).await?; + updated.mtime_nanos = st.mtime_nanos; + updated.atime_nanos = st.atime_nanos; + updated.ctime_nanos = now_nanos(); - /// Walk every part for this inode, take any in-flight receiver, await - /// it, and fold the etag into MpuState. Errors are returned to the - /// caller (the first one wins; we still drain the rest). - async fn drain_inflight_uploads(&self, handle: &FileHandle) -> FsResult<()> { - let final_size = *handle.size.read(); - if final_size == 0 { - return Ok(()); - } - let high = self - .config - .part_schedule - .locate(final_size - 1) - .ok_or(FsError::FileTooLarge)? - .part_index; - let mut receivers: Vec> = - Vec::new(); - for i in 0..=high { - let key = PartKey::new(handle.inode.id, i); - if let Some(part_arc) = self.pool.get(key) { - if let Some(rx) = part_arc.read().take_inflight() { - receivers.push(rx); - } - } - } - let mut first_err: Option = None; - for rx in receivers { - match rx.await { - Ok(Ok((part_index, etag))) => { - let mut g = handle.mpu.lock().await; - if let Some(state) = g.as_mut() { - state.record_part_uploaded(part_index, etag); - } - } - Ok(Err(e)) => { - if first_err.is_none() { - first_err = Some(e); - } - } - Err(_) => { - if first_err.is_none() { - first_err = Some(FsError::Io("flush worker dropped".into())); - } - } - } - } - if let Some(e) = first_err { - return Err(e); - } + txn.stage(updated); + txn.commit().await?; + st.dirty.clear(); + st.meta_dirty = false; Ok(()) } - async fn sync_via_single_put( - &self, - handle: &FileHandle, - key: &str, - final_size: u64, - dirty_parts: Vec<(u32, Arc>)>, - ) -> FsResult<()> { - debug_assert!( - dirty_parts.len() <= 1, - "small-file path implies ≤1 dirty part" - ); - - // Snapshot everything we need from the dirty part before any await. - // We capture the *dirty ranges* explicitly so we only overlay bytes - // the user actually wrote — `body[0..valid_len]` includes zero-fill - // bytes from `apply_write` extending past previous EOF, which would - // wrongly overwrite source bytes. - let part_snapshot: Option<(Bytes, bool, Vec>)> = - dirty_parts.first().map(|(_, part_arc)| { - let part = part_arc.read(); - let body = Bytes::copy_from_slice(part.body()); - let needs_source_load = !part.is_fully_dirty(); - let dirty_ranges = part.dirty_ranges().to_vec(); - (body, needs_source_load, dirty_ranges) - }); - let source_size = handle.inode.attrs.read().size; - - let body = if let Some((part_body, needs_source_load, dirty_ranges)) = part_snapshot { - let mut canvas = if needs_source_load && source_size > 0 { - let g = self.backend.get_blob(key, None).await?; - let mut buf = BytesMut::from(&g.body[..]); - if (buf.len() as u64) < final_size { - buf.resize(final_size as usize, 0); - } - buf - } else { - BytesMut::from(vec![0u8; final_size as usize].as_slice()) - }; - // Overlay only the dirty byte ranges from the buffer into the canvas. - for r in &dirty_ranges { - let start = r.start as usize; - let end = (r.end as usize).min(canvas.len()); - if start >= end { - continue; - } - canvas[start..end].copy_from_slice(&part_body[start..end]); - } - canvas.truncate(final_size as usize); - canvas.freeze() - } else { - let src = self.backend.get_blob(key, None).await?; - src.body - }; - - let meta = mpu::single_put_commit(&*self.backend, key, body).await?; - self.commit_book_keeping(handle, meta).await; - Ok(()) + pub async fn set_size(&self, handle: &FileHandle, new_size: u64) -> FsResult<()> { + if !handle.flags.write { + return Err(FsError::BadDescriptor); + } + let mut st = handle.state.lock().await; + st.size = new_size; + st.meta_dirty = true; + st.mtime_nanos = now_nanos(); + self.flush(handle.inode.objid(), &mut st).await } - async fn sync_via_mpu( + pub async fn set_times( &self, handle: &FileHandle, - key: &str, - dirty_parts: Vec<(u32, Arc>)>, + atime: Option, + mtime: Option, ) -> FsResult<()> { - // Lazy MPU begin if not yet done. - let mut mpu_guard = handle.mpu.lock().await; - if mpu_guard.is_none() { - let source_size = { - let a = handle.inode.attrs.read(); - if a.etag.is_empty() { - None - } else { - Some(a.size) - } - }; - let source_etag = { - let a = handle.inode.attrs.read(); - if a.etag.is_empty() { - None - } else { - Some(a.etag.clone()) - } - }; - let mut state = MpuState::new( - key.to_string(), - source_size, - source_etag, - self.config.part_schedule.clone(), - ); - let id = self - .backend - .multipart_begin(PutBlobInput { - key: key.to_string(), - body: Bytes::new(), - metadata: HashMap::new(), - content_type: None, - }) - .await?; - state.upload_id = Some(id); - *mpu_guard = Some(state); + // The same gate `pwrite` and `set_size` apply, and it was missing here. + // A timestamp is metadata, but setting one is still a *commit*: `flush` + // publishes a new signed root record under Object Lock retention. A + // handle opened read-only that can advance the anchor chain is not + // read-only in any sense worth the name. + // + // POSIX would settle this by ownership rather than by the descriptor's + // open mode — `futimens` on an `O_RDONLY` fd is legal for the owner. + // There is no owner here: [`Attrs`] carries `mode` but no uid, so + // "are you allowed?" has no answer other than what the handle was + // opened for. `set_times_at` stays ungated for the same reason, having + // no handle to ask. + if !handle.flags.write { + return Err(FsError::BadDescriptor); } - // Snapshot what the parallel uploaders need from state, then drop - // the guard so we don't hold it across the upload window. The caller - // contract is that no other task races sync() on the same handle, so - // the upload_id and source_size remain valid until we re-acquire. - let (upload_id, source_size) = { - let state = mpu_guard.as_ref().expect("MPU started"); - ( - state.upload_id.clone().expect("upload begun"), - state.source_size.unwrap_or(0), - ) - }; - drop(mpu_guard); - - // Parallel UploadPart pass. Capture final_size *before* the await so - // each task can clamp its part length correctly. - let final_size_for_upload = *handle.size.read(); - let upload_results = self - .upload_parts_concurrent( - key, - &upload_id, - source_size, - final_size_for_upload, - dirty_parts, - self.config.max_parallel_parts, - ) - .await; - - // Re-acquire MpuState and record results (sequentially, but cheap). - let mut mpu_guard = handle.mpu.lock().await; - let state = mpu_guard.as_mut().expect("MPU started"); - for r in upload_results { - let (part_index, etag) = r?; - state.record_part_uploaded(part_index, etag); + let mut st = handle.state.lock().await; + if let Some(t) = atime { + st.atime_nanos = to_nanos(t); } - - // Make sure high_water includes the trailing parts of the file even - // if they weren't dirty (so copy_plan emits UploadPartCopy for them). - let final_size = *handle.size.read(); - if final_size > 0 { - let high = self - .config - .part_schedule - .locate(final_size - 1) - .ok_or(FsError::FileTooLarge)? - .part_index as i32; - if high > state.high_water { - state.high_water = high; - } + if let Some(t) = mtime { + st.mtime_nanos = to_nanos(t); } - - let meta = mpu::commit( - state, - self.backend.clone(), - self.config.max_merge_copy_bytes, - self.config.max_parallel_copy, - key, - ) - .await?; - // After commit, drop the MpuState's upload_id (commit clears it). - drop(mpu_guard); - self.commit_book_keeping(handle, meta).await; - Ok(()) - } - - /// Common post-sync cleanup: refresh inode attrs, mark clean parts, - /// release pool memory. - async fn commit_book_keeping(&self, handle: &FileHandle, meta: BlobMeta) { - // Update inode attrs to reflect the commit. - *handle.inode.attrs.write() = attrs_from_meta(&meta); - *handle.size.write() = meta.size; - handle.inode.set_state(InodeState::Cached); - - // Walk known parts in pool and clear dirty/flushed → clean. - // (We don't have an iterator over per-inode parts on the pool, so - // we just forget them — they can be re-fetched if needed.) - self.pool.forget_inode(handle.inode.id); + st.meta_dirty = true; + self.flush(handle.inode.objid(), &mut st).await } - // ------------------- buffer-pool helpers ------------------- - - /// Concurrently upload every dirty part. Each task does: snapshot → - /// materialize (with optional RMW GET for partial parts) → mark_flushing - /// → UploadPart → mark_flushed / mark_flush_failed → return etag. - /// - /// Returns one result per input part, in completion order. The caller is - /// responsible for recording successful etags into `MpuState`. - async fn upload_parts_concurrent( + pub async fn set_times_at( &self, - key: &str, - upload_id: &crate::backend::MultipartId, - source_size: u64, - final_size: u64, - dirty_parts: Vec<(u32, Arc>)>, - max_parallel: usize, - ) -> Vec> { - use futures::stream::{self, StreamExt}; - - let backend = self.backend.clone(); - let key = Arc::new(key.to_string()); - let upload_id = Arc::new(upload_id.clone()); - let schedule = self.config.part_schedule.clone(); - let cap = max_parallel.max(1); - let semaphore = self.flusher.semaphore(); - - stream::iter(dirty_parts.into_iter().map(|(part_index, part_arc)| { - let backend = backend.clone(); - let key = key.clone(); - let upload_id = upload_id.clone(); - let schedule = schedule.clone(); - let semaphore = semaphore.clone(); - async move { - // Acquire a permit from the per-Fs flusher BEFORE doing any - // backend I/O. This caps total concurrent uploads across all - // open files, not just within one sync() call. - let _permit = semaphore - .acquire_owned() - .await - .expect("semaphore never closed"); - upload_one_part( - backend, - key.as_str(), - upload_id.as_ref(), - part_index, - part_arc, - &schedule, - source_size, - final_size, - ) - .await - } - })) - .buffer_unordered(cap) - .collect() - .await - } + base: &Arc, + path: &str, + atime: Option, + mtime: Option, + follow: bool, + ) -> FsResult<()> { + let ino = self.resolve(base, path, follow).await?; + let mut txn = self.store.begin().await?; + let blocks = txn.blocks().clone(); + let objset = txn.objset().clone(); - /// Fetch (or create) a part for read access. If absent from the pool - /// AND the part lies within the source object, do a ranged GET; else - /// return an empty PartBuf. - async fn get_or_fetch_part( - &self, - handle: &FileHandle, - part_index: u32, - ) -> FsResult>> { - let key = PartKey::new(handle.inode.id, part_index); - if let Some(p) = self.pool.get(key) { - return Ok(p); + let mut d = objset.get_allocated(&blocks, ino.objid()).await?; + if let Some(t) = atime { + d.atime_nanos = to_nanos(t); } - let part_range = self - .config - .part_schedule - .part_range(part_index) - .ok_or(FsError::FileTooLarge)?; - let source_size = handle.inode.attrs.read().size; - if part_range.start >= source_size { - // Empty part past EOF. - let part = PartBuf::new_empty_dirty( - part_index, - part_range.end - part_range.start, - part_range.start, - ); - return Ok(self.pool.insert(key, part)); + if let Some(t) = mtime { + d.mtime_nanos = to_nanos(t); } - let r = part_range.start..part_range.end.min(source_size); - let g = self - .backend - .get_blob(&self.tree.current_s3_key(&handle.inode), Some(r)) - .await?; - let part = PartBuf::new_clean( - part_index, - part_range.end - part_range.start, - part_range.start, - g.body, - ); - Ok(self.pool.insert(key, part)) - } - - /// Like `get_or_fetch_part` but used for the write path. Identical logic - /// today; named separately so future write-bypass optimisations have a - /// hook. - async fn get_or_fetch_part_for_write( - &self, - handle: &FileHandle, - part_index: u32, - ) -> FsResult>> { - self.get_or_fetch_part(handle, part_index).await + d.ctime_nanos = now_nanos(); + txn.stage(d); + txn.commit().await.map(|_| ()) } } -/// One-shot helper for `upload_parts_concurrent`: snapshot, materialize -/// (with RMW for partial parts), mark Flushing, `UploadPart`, mark Flushed. -/// Errors transition the part back to Dirty so a retry can pick it up. -#[allow(clippy::too_many_arguments)] -pub(crate) async fn upload_one_part( - backend: Arc, - key: &str, - upload_id: &crate::backend::MultipartId, - part_index: u32, - part_arc: Arc>, - schedule: &crate::config::PartSchedule, - source_size: u64, - // `final_size`: file size *after* this sync — used to compute the part's - // actual byte length (so a sub-part dirty write to the middle of a large - // file doesn't truncate the unchanged tail). - final_size: u64, -) -> FsResult<(u32, String)> { - // Snapshot AND mark_flushing in one critical section so that any - // concurrent pwrite is either fully captured by the snapshot OR sees - // state=Flushing and bails (returns WouldBlock; pwrite handles by - // awaiting the in-flight receiver). Without this, a write that lands - // between snapshot and mark_flushing is lost when mark_flushed clears - // the dirty bitmap. - // - // If state is already Flushing (because eager pwrite raced with us - // and snapshot+marked first), skip the transition. - let (part_body, fully_dirty, dirty_ranges_owned) = { - let mut part = part_arc.write(); - let body = Bytes::copy_from_slice(part.body()); - let fully = part.is_fully_dirty(); - let ranges = part.dirty_ranges().to_vec(); - if matches!(part.state, crate::buffer::PartState::Dirty) { - // Normal path: transition Dirty → Flushing here. - part.state = crate::buffer::PartState::Flushing; - } - (body, fully, ranges) - }; - - let part_range = schedule - .part_range(part_index) - .ok_or(FsError::FileTooLarge)?; - // The part's actual size in the destination file: full part_size, except - // possibly clipped if this is the last part. - let part_actual_size = (final_size.saturating_sub(part_range.start)) - .min(part_range.end - part_range.start) as usize; +// -- snapshots --------------------------------------------------------------- - let mut materialized = if fully_dirty && part_body.len() == part_actual_size { - BytesMut::from(&part_body[..]) - } else { - // RMW: load source bytes for this part's range, then overlay only - // the dirty subranges from the buffer. - let source_end = part_range.end.min(source_size); - let source_start = part_range.start; - let loaded = if source_end > source_start { - let g = backend - .get_blob(key, Some(source_start..source_end)) - .await?; - g.body - } else { - Bytes::new() - }; - let mut canvas = BytesMut::from(&loaded[..]); - if canvas.len() < part_actual_size { - canvas.resize(part_actual_size, 0); - } - for r in &dirty_ranges_owned { - let start = r.start as usize; - let end = (r.end as usize).min(canvas.len()); - if start >= end { - continue; - } - canvas[start..end].copy_from_slice(&part_body[start..end]); - } - canvas - }; - materialized.truncate(part_actual_size); - - let body = materialized.freeze(); - let r = backend - .multipart_upload_part(key, upload_id, part_index + 1, body) - .await; - let etag = match r { - Ok(o) => o.e_tag, - Err(e) => { - let mut p = part_arc.write(); - let _ = p.mark_flush_failed(); - return Err(e); - } - }; - { - let mut p = part_arc.write(); - p.mark_flushed(etag.clone())?; +impl Fs { + /// The newest `limit` committed states, most recent first. + /// + /// Every root record is a snapshot. Copy-on-write means the blocks a past + /// root names were never overwritten, so taking one costs nothing and + /// keeping one costs only whatever storage its blocks already occupy. + pub async fn snapshots(&self, limit: usize) -> FsResult> { + Ok(self + .store + .list_snapshots(limit) + .await? + .into_iter() + .map(|r| SnapshotInfo { + seq: r.seq, + txg: r.txg, + timestamp: crate::inode::from_nanos(r.timestamp_nanos), + merkle_root: *r.merkle_root(), + }) + .collect()) + } + + /// Open a past state for reading. + pub async fn open_snapshot(&self, seq: u64) -> FsResult { + let snapshot = self.store.snapshot(seq).await?; + Ok(SnapshotFs { + objset: snapshot.objset().clone(), + blocks: self.blocks(), + config: self.config.clone(), + info: SnapshotInfo { + seq: snapshot.root.seq, + txg: snapshot.root.txg, + timestamp: crate::inode::from_nanos(snapshot.root.timestamp_nanos), + merkle_root: *snapshot.root.merkle_root(), + }, + }) } - Ok((part_index, etag)) } -fn attrs_from_meta(meta: &BlobMeta) -> Attrs { - Attrs { - size: meta.size, - etag: meta.e_tag.clone(), - last_modified: meta.last_modified, - content_type: meta.content_type.clone(), - metadata: meta.metadata.clone(), - fetched_at: std::time::Instant::now(), - } +/// Identifying details of one committed state. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct SnapshotInfo { + /// Root sequence number. Also its key in the roots bucket. + pub seq: u64, + pub txg: u64, + pub timestamp: SystemTime, + /// The hash covering the entire filesystem at this point. + pub merkle_root: crate::crypto::Hash256, } -#[cfg(test)] -mod tests { - use super::*; - use crate::backend::memory::MemoryBackend; - - fn fresh_fs(memory_limit: u64, single_part: u64) -> (Arc, Arc) { - let backend = Arc::new(MemoryBackend::new()); - let cfg = Config::builder() - .memory_limit_bytes(memory_limit) - .single_part_threshold(single_part) - .build(); - let fs = Fs::new(backend.clone() as Arc, Arc::new(cfg)); - (backend, fs) - } - - fn small_fs() -> (Arc, Arc) { - // Default-ish: 64 MiB pool, 5 MiB single-part threshold. - fresh_fs(64 * 1024 * 1024, 5 * 1024 * 1024) - } - - fn tiny_fs() -> (Arc, Arc) { - // Tiny part schedule: 4 bytes/part × 100, 8-byte single-part threshold. - let backend = Arc::new(MemoryBackend::new()); - let cfg = Config::builder() - .part_schedule(crate::config::PartSchedule { - tiers: vec![(4, 100)], - }) - .single_part_threshold(8) - .max_merge_copy_bytes(1024 * 1024) - .memory_limit_bytes(1024 * 1024) - .build(); - let fs = Fs::new(backend.clone() as Arc, Arc::new(cfg)); - (backend, fs) - } - - // ---------- mkdir / unlink / rmdir ---------- - - #[tokio::test] - async fn mkdir_creates_explicit_marker_and_appears_in_listing() { - let (backend, fs) = small_fs(); - let root = fs.root(); - let dir = fs.mkdir(&root, "subdir").await.unwrap(); - assert!(dir.is_dir()); - // Explicit marker key exists at "subdir/". - assert!(backend.head_blob("subdir/").await.is_ok()); - // It's enumerable from root. - let snap = fs.read_dir(&root).await.unwrap(); - assert!(snap.iter().any(|e| e.name == "subdir")); - } - - #[tokio::test] - async fn mkdir_already_exists_errors() { - let (_b, fs) = small_fs(); - let root = fs.root(); - fs.mkdir(&root, "x").await.unwrap(); - assert!(matches!( - fs.mkdir(&root, "x").await, - Err(FsError::AlreadyExists) - )); - } - - #[tokio::test] - async fn unlink_removes_object_and_inode() { - let (backend, fs) = small_fs(); - let root = fs.root(); - // Drop a file directly via the backend, then look it up. - backend - .put_blob(PutBlobInput { - key: "f.txt".into(), - body: Bytes::from_static(b"x"), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - let _ = fs.tree.lookup(&root, "f.txt").await.unwrap(); - fs.unlink(&root, "f.txt").await.unwrap(); - assert!(matches!( - backend.head_blob("f.txt").await, - Err(FsError::NotFound) - )); - assert!(matches!( - fs.tree.lookup(&root, "f.txt").await, - Err(FsError::NotFound) - )); - } - - #[tokio::test] - async fn unlink_on_directory_errors() { - let (_b, fs) = small_fs(); - let root = fs.root(); - fs.mkdir(&root, "d").await.unwrap(); - assert!(matches!( - fs.unlink(&root, "d").await, - Err(FsError::IsDirectory) - )); - } - - #[tokio::test] - async fn rmdir_empty_succeeds() { - let (backend, fs) = small_fs(); - let root = fs.root(); - fs.mkdir(&root, "empty").await.unwrap(); - fs.rmdir(&root, "empty").await.unwrap(); - assert!(matches!( - backend.head_blob("empty/").await, - Err(FsError::NotFound) - )); - } +/// A past state of the filesystem, opened read-only. +/// +/// Reads verify exactly as the live mount does — same checksums, same +/// position-binding — because a snapshot is not a copy of anything. It is the +/// same blocks, reached through an older root. +#[derive(Debug)] +pub struct SnapshotFs { + objset: ObjectSet, + blocks: Arc, + config: Arc, + info: SnapshotInfo, +} - #[tokio::test] - async fn rmdir_non_empty_returns_not_empty() { - let (backend, fs) = small_fs(); - let root = fs.root(); - fs.mkdir(&root, "d").await.unwrap(); - backend - .put_blob(PutBlobInput { - key: "d/inner".into(), - body: Bytes::from_static(b"x"), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - assert!(matches!(fs.rmdir(&root, "d").await, Err(FsError::NotEmpty))); +impl SnapshotFs { + pub fn info(&self) -> SnapshotInfo { + self.info } - // ---------- open / O_CREAT / O_EXCL / O_TRUNC ---------- - - #[tokio::test] - async fn open_create_writes_empty_file() { - let (backend, fs) = small_fs(); - let h = fs - .open( - "hello.txt", - OpenFlags { - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - assert_eq!(*h.size.read(), 0); - assert!(backend.head_blob("hello.txt").await.is_ok()); - fs.close(&h).await.unwrap(); + pub fn root(&self) -> Arc { + Inode::new(ROOT_OBJID, 0) + } + + pub async fn lookup(&self, base: &Arc, path: &str) -> FsResult> { + let objid = resolve_in( + &self.objset, + &self.blocks, + // A snapshot is read-only and whole: there is no tenant to confine + // to, and confining to one would hide the rest of a snapshot from + // the operator reading it. + ROOT_OBJID, + base.objid(), + path, + true, + self.config.max_symlink_depth, + ) + .await?; + let d = self.objset.get_allocated(&self.blocks, objid).await?; + Ok(Inode::new(objid, d.gen)) } - #[tokio::test] - async fn open_create_excl_collides_returns_already_exists() { - let (_b, fs) = small_fs(); - let _h1 = fs.open("k", OpenFlags::create_new()).await.unwrap(); - let r = fs.open("k", OpenFlags::create_new()).await; - assert!(matches!(r, Err(FsError::AlreadyExists))); + pub async fn stat(&self, ino: &Arc) -> FsResult { + Attrs::from_dnode(&self.objset.get_allocated(&self.blocks, ino.objid()).await?) } - #[tokio::test] - async fn open_truncate_zeroes_existing_file() { - let (backend, fs) = small_fs(); - backend - .put_blob(PutBlobInput { - key: "k".into(), - body: Bytes::from_static(b"some content"), - metadata: HashMap::new(), - content_type: None, + pub async fn read_dir(&self, dir: &Arc) -> FsResult> { + let d = self.objset.get_allocated(&self.blocks, dir.objid()).await?; + require_dir(&d)?; + DirTxn::load(&self.blocks, &d) + .await? + .list() + .await? + .into_iter() + .map(|e| { + Ok(DirEntry { + name: e.name, + kind: InodeKind::from_dnode(e.kind)?, + objid: e.objid, + }) }) - .await - .unwrap(); - let h = fs - .open( - "k", - OpenFlags { - read: true, - write: true, - truncate: true, - ..Default::default() - }, - ) - .await - .unwrap(); - assert_eq!(*h.size.read(), 0); - let g = backend.get_blob("k", None).await.unwrap(); - assert!(g.body.is_empty()); - fs.close(&h).await.unwrap(); - } - - #[tokio::test] - async fn open_no_read_no_write_errors() { - let (_b, fs) = small_fs(); - let r = fs.open("x", OpenFlags::default()).await; - assert!(matches!(r, Err(FsError::Invalid(_)))); - } - - // ---------- pread / pwrite / sync ---------- - - #[tokio::test] - async fn write_then_sync_then_read_small_file() { - let (backend, fs) = small_fs(); - let h = fs - .open( - "small.txt", - OpenFlags { - read: true, - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - let n = fs.pwrite(&h, 0, b"hello world").await.unwrap(); - assert_eq!(n, 11); - fs.sync(&h).await.unwrap(); - // Verify in S3. - let g = backend.get_blob("small.txt", None).await.unwrap(); - assert_eq!(&g.body[..], b"hello world"); - // Re-open and read. - let h2 = fs.open("small.txt", OpenFlags::read_only()).await.unwrap(); - let body = fs.pread(&h2, 0, 100).await.unwrap(); - assert_eq!(&body[..], b"hello world"); - fs.close(&h).await.unwrap(); - fs.close(&h2).await.unwrap(); + .collect() } - #[tokio::test] - async fn append_to_existing_via_random_write_uses_mpu_path() { - let (backend, fs) = tiny_fs(); - // Pre-populate a 16-byte source object (4 parts of 4 bytes each). - backend - .put_blob(PutBlobInput { - key: "big".into(), - body: Bytes::from_static(b"AAAABBBBCCCCDDDD"), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - let h = fs - .open( - "big", - OpenFlags { - read: true, - write: true, - ..Default::default() - }, - ) - .await - .unwrap(); - // Modify part 1 (bytes 4..8): "BBBB" → "ZZZZ". - fs.pwrite(&h, 4, b"ZZZZ").await.unwrap(); - // Extend file by appending at offset 16: 4 new bytes "EEEE" (part 4). - fs.pwrite(&h, 16, b"EEEE").await.unwrap(); - fs.sync(&h).await.unwrap(); - let g = backend.get_blob("big", None).await.unwrap(); - assert_eq!(&g.body[..], b"AAAAZZZZCCCCDDDDEEEE"); - // No orphan MPU. - assert_eq!(backend.mpu_count(), 0); - fs.close(&h).await.unwrap(); - } + /// Read a byte range of a file as it was at this snapshot. + pub async fn read(&self, ino: &Arc, offset: u64, len: usize) -> FsResult { + let d = self.objset.get_allocated(&self.blocks, ino.objid()).await?; + if d.kind == DnodeKind::Dir { + return Err(FsError::IsDirectory); + } + if offset >= d.size || len == 0 { + return Ok(Bytes::new()); + } + let end = offset.saturating_add(len as u64).min(d.size); + let record = d.record_size() as u64; - #[tokio::test] - async fn read_after_partial_write_sees_the_overlay() { - let (backend, fs) = tiny_fs(); - backend - .put_blob(PutBlobInput { - key: "k".into(), - body: Bytes::from_static(b"AAAABBBBCCCC"), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - let h = fs - .open( - "k", - OpenFlags { - read: true, - write: true, - ..Default::default() - }, - ) - .await - .unwrap(); - fs.pwrite(&h, 5, b"XX").await.unwrap(); - // Read should see the overlay even before sync (via buffer pool). - let body = fs.pread(&h, 0, 12).await.unwrap(); - assert_eq!(&body[..], b"AAAABXXBCCCC"); - fs.close(&h).await.unwrap(); + let mut out = Vec::with_capacity((end - offset) as usize); + let mut pos = offset; + while pos < end { + let index = pos / record; + let within = (pos % record) as usize; + let take = ((record - within as u64).min(end - pos)) as usize; + let block = read_data_block(&self.blocks, &d, index).await?; + let slice = block.get(within..).unwrap_or(&[]); + let n = take.min(slice.len()); + out.extend_from_slice(&slice[..n]); + out.resize(out.len() + (take - n), 0); + pos += take as u64; + } + Ok(Bytes::from(out)) } +} - #[tokio::test] - async fn small_file_inplace_edit_preserves_unchanged_source_bytes() { - // Regression: the small-file `sync_via_single_put` used to overlay - // `body[0..valid_len]` onto the canvas, which included zero-fill bytes - // from `apply_write` extending past previous EOF — wrongly clobbering - // source bytes. SQLite blocker. - let (backend, fs) = small_fs(); - backend - .put_blob(PutBlobInput { - key: "db".into(), - body: Bytes::from_static(b"AAAABBBBCCCCDDDD"), // 16 bytes - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - let h = fs - .open( - "db", - OpenFlags { - read: true, - write: true, - ..Default::default() - }, - ) - .await - .unwrap(); - // Modify only bytes 8..10. Bytes 0..8 must come from source, not zeros. - fs.pwrite(&h, 8, b"ZZ").await.unwrap(); - fs.sync(&h).await.unwrap(); - fs.close(&h).await.unwrap(); +// -- helpers ---------------------------------------------------------------- - let g = backend.get_blob("db", None).await.unwrap(); - assert_eq!(&g.body[..], b"AAAABBBBZZCCDDDD"); +fn require_dir(d: &Dnode) -> FsResult<()> { + if d.kind == DnodeKind::Dir { + Ok(()) + } else { + Err(FsError::NotDirectory) } +} - #[tokio::test] - async fn large_file_subpart_edit_preserves_unchanged_part_tail() { - // Regression: in `upload_one_part`, materialized was truncated to - // `valid_len` instead of `min(part_size, file_size - part_start)`. - // For a file > threshold with sub-part dirty content in a non-last - // part, this shrank the part body and corrupted the file. SQLite - // blocker for any DB > 5 MiB. - let backend = Arc::new(MemoryBackend::new()); - let cfg = Config::builder() - .part_schedule(crate::config::PartSchedule { - tiers: vec![(8, 100)], - }) - .single_part_threshold(8) // anything above 8 bytes uses MPU - .max_parallel_parts(2) - .max_parallel_copy(2) - .max_merge_copy_bytes(1024) - .memory_limit_bytes(1024 * 1024) - .build(); - let fs = Fs::new(backend.clone() as Arc, Arc::new(cfg)); - - // Source: 24 bytes = 3 parts of 8 bytes. - backend - .put_blob(PutBlobInput { - key: "k".into(), - body: Bytes::from_static(b"AAAAAAAABBBBBBBBCCCCCCCC"), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - - let h = fs - .open( - "k", - OpenFlags { - read: true, - write: true, - ..Default::default() - }, - ) - .await - .unwrap(); - // Write 2 bytes at offset 8 (part 1, offset 0): modifies first 2 - // bytes of "BBBBBBBB". The rest of part 1 ("BBBBBB") must survive. - fs.pwrite(&h, 8, b"ZZ").await.unwrap(); - fs.sync(&h).await.unwrap(); - fs.close(&h).await.unwrap(); +fn touch(d: &mut Dnode, now: u64) { + d.mtime_nanos = now; + d.ctime_nanos = now; +} - let g = backend.get_blob("k", None).await.unwrap(); - assert_eq!(g.body.len(), 24, "file size unchanged"); - assert_eq!(&g.body[..], b"AAAAAAAAZZBBBBBBCCCCCCCC"); - assert_eq!(backend.mpu_count(), 0); +/// Split a path into its parent portion and final component. +fn split_last(path: &str) -> FsResult<(&str, &str)> { + let trimmed = path.trim_end_matches('/'); + if trimmed.is_empty() { + return Err(FsError::Invalid("empty path")); } + Ok(match trimmed.rsplit_once('/') { + Some((parent, name)) => (parent, name), + None => ("", trimmed), + }) +} - #[tokio::test] - async fn close_without_sync_aborts_in_flight_mpu() { - let (backend, fs) = tiny_fs(); - backend - .put_blob(PutBlobInput { - key: "k".into(), - body: Bytes::from_static(b"AAAABBBBCCCC"), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - let h = fs - .open( - "k", - OpenFlags { - read: true, - write: true, - ..Default::default() - }, - ) - .await - .unwrap(); - // Trigger MPU lazily by writing across part boundaries. - fs.pwrite(&h, 0, b"ZZZZZZZZZZZZ").await.unwrap(); - // Manually start an MPU by calling sync once... but we want to - // verify abort, so simulate: start MPU, then close without sync. - // For this, we directly call the lazy path by performing sync_via_mpu's - // begin-step indirectly via sync, then truncate state and close. - // Easier: just call close — sync wasn't called, so even if MPU was - // begun, abort should clean it up. We don't actually start one in - // this code path because writes don't begin MPU lazily — only sync - // does. So this verifies the no-MPU close path. - fs.close(&h).await.unwrap(); - assert_eq!(backend.mpu_count(), 0); - } - - // ---------- rename ---------- - - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn rename_file_async_copy_delete() { - let (backend, fs) = small_fs(); - let root = fs.root(); - let h = fs - .open( - "old", - OpenFlags { - read: true, - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - fs.pwrite(&h, 0, b"data").await.unwrap(); - fs.sync(&h).await.unwrap(); - fs.close(&h).await.unwrap(); - - fs.rename(&root, "old", &root, "new").await.unwrap(); - // Wait for the background worker to finish copy+delete. - let new_ino = fs.tree.lookup(&root, "new").await.unwrap(); - fs.wait_for_rename(&new_ino).await.unwrap(); - - assert!(matches!( - backend.head_blob("old").await, - Err(FsError::NotFound) - )); - let g = backend.get_blob("new", None).await.unwrap(); - assert_eq!(&g.body[..], b"data"); - } - - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn rename_directory_recursive_moves_all_descendants() { - let (backend, fs) = small_fs(); - let root = fs.root(); - // Build /old/{a.txt, b.txt, sub/c.txt} - fs.mkdir(&root, "old").await.unwrap(); - for (path, body) in [ - ("old/a.txt", b"AAA" as &[u8]), - ("old/b.txt", b"BBB"), - ("old/sub/c.txt", b"CCC"), - ] { - backend - .put_blob(PutBlobInput { - key: path.into(), - body: Bytes::copy_from_slice(body), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - } - // Force the inode tree to know about /old. - let _ = fs.read_dir(&root).await.unwrap(); - - fs.rename(&root, "old", &root, "new").await.unwrap(); - let new_ino = fs.tree.lookup(&root, "new").await.unwrap(); - fs.wait_for_rename(&new_ino).await.unwrap(); - - // Old keys are gone; new keys carry the content. - for k in ["old/a.txt", "old/b.txt", "old/sub/c.txt", "old/"] { - assert!( - matches!(backend.head_blob(k).await, Err(FsError::NotFound)), - "leftover {k}" - ); - } - for (k, expected) in [ - ("new/a.txt", b"AAA" as &[u8]), - ("new/b.txt", b"BBB"), - ("new/sub/c.txt", b"CCC"), - ] { - let g = backend.get_blob(k, None).await.unwrap(); - assert_eq!(&g.body[..], expected, "key {k}"); +async fn read_symlink(blocks: &BlockStore, d: &Dnode) -> FsResult { + let bytes = if !d.inline.is_empty() { + d.inline.clone() + } else { + let mut out = Vec::with_capacity(d.size as usize); + for i in 0..d.block_count() { + out.extend_from_slice(&read_data_block(blocks, d, i).await?); } - } + out + }; + String::from_utf8(bytes).map_err(|_| FsError::IllegalByteSequence) +} - #[tokio::test] - async fn rename_directory_into_self_rejected() { - let (_b, fs) = small_fs(); - let root = fs.root(); - fs.mkdir(&root, "d").await.unwrap(); - let d = fs.tree.lookup(&root, "d").await.unwrap(); - let r = fs.rename(&root, "d", &d, "inner").await; - assert!(matches!(r, Err(FsError::Invalid(_)))); - } - - /// Right after `rename` returns and BEFORE the worker completes, - /// reads against the new path resolve via `current_s3_key` to the OLD - /// key, which still exists. The worker eventually catches up. - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn rename_destination_reads_old_key_until_worker_finishes() { - let (backend, fs) = small_fs(); - let root = fs.root(); - // Seed an existing object so the OLD key has content. - backend - .put_blob(PutBlobInput { - key: "src.bin".into(), - body: Bytes::from_static(b"hello"), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - let _ = fs.read_dir(&root).await.unwrap(); - - fs.rename(&root, "src.bin", &root, "dst.bin").await.unwrap(); - let new_ino = fs.tree.lookup(&root, "dst.bin").await.unwrap(); - // current_s3_key still points at the old key while rename in flight. - let key = fs.tree.current_s3_key(&new_ino); - assert!( - key == "src.bin" || key == "dst.bin", - "key should be either old (rename in flight) or new (worker completed): {key}" - ); - - fs.wait_for_rename(&new_ino).await.unwrap(); - assert_eq!(fs.tree.current_s3_key(&new_ino), "dst.bin"); - let g = backend.get_blob("dst.bin", None).await.unwrap(); - assert_eq!(&g.body[..], b"hello"); - } - - /// A second `rename` of the same inode while the first is still in - /// flight is rejected with `WouldBlock`. - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn second_rename_while_in_flight_returns_wouldblock() { - let (backend, fs) = small_fs(); - let root = fs.root(); - backend - .put_blob(PutBlobInput { - key: "f".into(), - body: Bytes::from_static(b"x"), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - let _ = fs.read_dir(&root).await.unwrap(); - fs.rename(&root, "f", &root, "g").await.unwrap(); - // Second rename of the same inode under its NEW name. The - // worker may or may not have run yet. If state is still set, - // expect WouldBlock; if it cleared, expect Ok. - let g_ino = fs.tree.lookup(&root, "g").await.unwrap(); - let r = fs.rename(&root, "g", &root, "h").await; - if g_ino.rename_state().is_some() { - assert!(matches!(r, Err(FsError::WouldBlock))); - } else { - r.unwrap(); +/// Walk `path` from `start`, following symlinks. +/// +/// Recursion is bounded by `depth`, which is decremented on every symlink hop +/// rather than every component, so a chain of links cannot outlast it. +/// Walk `path`, never leaving the subtree rooted at `scope`. +/// +/// `scope` is what makes this a capability rather than a convention. Every way +/// a path can name something above where it started is redirected to `scope` +/// instead of to the filesystem root: +/// +/// - an **absolute path** starts at `scope`, so `/etc/passwd` means +/// `/etc/passwd` and cannot mean anything else; +/// - **`..`** at `scope` stays at `scope`, so no number of them walks out; +/// - an **absolute symlink target** resolves from `scope` too, so a guest +/// cannot manufacture an escape by writing one. +/// +/// Those three are the whole attack surface, and they hold inductively: a walk +/// begins at `scope` or below, and none of the three can take it higher. +/// +/// Pass [`ROOT_OBJID`] for an unscoped filesystem, which is what a guest with +/// the whole store as its preopen gets. +async fn resolve_in( + objset: &ObjectSet, + blocks: &BlockStore, + scope: u64, + start: u64, + path: &str, + follow_final: bool, + depth: u32, +) -> FsResult { + if depth == 0 { + return Err(FsError::Loop); + } + let mut current = if path.starts_with('/') { scope } else { start }; + + let components: Vec<&str> = path + .split('/') + .filter(|c| !c.is_empty() && *c != ".") + .collect(); + + for (i, comp) in components.iter().enumerate() { + let last = i + 1 == components.len(); + + if *comp == ".." { + // Clamped at the scope, not at the mount. Without this a tenant + // holding `/tenants/a` would reach `/tenants` — and from there + // every other tenant — with two components of ordinary path. + if current == scope { + continue; + } + let d = objset.get_allocated(blocks, current).await?; + // The root is its own parent, so `..` stops at the mount rather + // than escaping it. + current = d.parent_objid; + continue; + } + validate_segment(comp)?; + + let dir_dnode = objset.get_allocated(blocks, current).await?; + require_dir(&dir_dnode)?; + let mut dir = DirTxn::load(blocks, &dir_dnode).await?; + let entry = dir.lookup(comp).await?.ok_or(FsError::NotFound)?; + current = entry.objid; + + if entry.kind == DnodeKind::Symlink && (!last || follow_final) { + let link = objset.get_allocated(blocks, current).await?; + let target = read_symlink(blocks, &link).await?; + let from = if target.starts_with('/') { + scope + } else { + dir_dnode.objid + }; + current = Box::pin(resolve_in( + objset, + blocks, + scope, + from, + &target, + true, + depth - 1, + )) + .await?; } - // Drain the queue to settle. - let _ = fs.wait_for_rename(&g_ino).await; } + Ok(current) +} - // ---------- set_size ---------- - - #[tokio::test] - async fn set_size_truncate_to_zero() { - let (backend, fs) = small_fs(); - backend - .put_blob(PutBlobInput { - key: "k".into(), - body: Bytes::from_static(b"some content"), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - let h = fs - .open( - "k", - OpenFlags { - read: true, - write: true, - ..Default::default() - }, - ) - .await - .unwrap(); - fs.set_size(&h, 0).await.unwrap(); - assert_eq!(*h.size.read(), 0); - let g = backend.get_blob("k", None).await.unwrap(); - assert!(g.body.is_empty()); - fs.close(&h).await.unwrap(); - } +#[cfg(test)] +mod tests; - #[tokio::test] - async fn set_size_shrink_partial_keeps_prefix() { - let (backend, fs) = small_fs(); - backend - .put_blob(PutBlobInput { - key: "k".into(), - body: Bytes::from_static(b"abcdefghij"), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - let h = fs - .open( - "k", - OpenFlags { - read: true, - write: true, - ..Default::default() - }, - ) - .await - .unwrap(); - fs.set_size(&h, 4).await.unwrap(); - let g = backend.get_blob("k", None).await.unwrap(); - assert_eq!(&g.body[..], b"abcd"); - } +#[cfg(test)] +mod scope_tests { + use super::*; + use crate::backend::memory::MemoryBackend; + use crate::config::Config; + use crate::crypto::MasterSecret; - #[tokio::test] - async fn set_size_grow_zero_fills_via_pwrite() { - let (backend, fs) = small_fs(); - let h = fs - .open( - "k", - OpenFlags { - read: true, - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - fs.pwrite(&h, 0, b"abc").await.unwrap(); - fs.set_size(&h, 10).await.unwrap(); - fs.sync(&h).await.unwrap(); - let g = backend.get_blob("k", None).await.unwrap(); - assert_eq!(&g.body[..], b"abc\0\0\0\0\0\0\0"); - } + /// Two tenants under `/tenants`, plus a secret at the root that neither + /// should ever reach. + async fn tenanted() -> (Arc, Arc, Arc) { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([3u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .unwrap(); - #[tokio::test] - async fn set_size_grow_too_large_rejected() { - let (_b, fs) = small_fs(); - let h = fs - .open( - "k", - OpenFlags { - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - let r = fs.set_size(&h, 200 * 1024 * 1024).await; - assert!(matches!(r, Err(FsError::Invalid(_)))); + let root = fs.root(); + fs.mkdir(&root, "runtime-only").await.unwrap(); + let tenants = fs.mkdir(&root, "tenants").await.unwrap(); + let alice = fs.mkdir(&tenants, "alice").await.unwrap(); + let bob = fs.mkdir(&tenants, "bob").await.unwrap(); + fs.mkdir(&bob, "bobs-private").await.unwrap(); + (fs, alice, bob) } - // ---------- set_times ---------- - + /// Inside her own subtree Alice works normally. #[tokio::test] - async fn set_times_persists_mtime_to_metadata() { - let (backend, fs) = small_fs(); - backend - .put_blob(PutBlobInput { - key: "k".into(), - body: Bytes::from_static(b"x"), - metadata: HashMap::new(), - content_type: None, - }) + async fn a_scope_does_not_restrict_what_is_inside_it() { + let (fs, alice, _) = tenanted().await; + fs.mkdir(&alice, "work").await.unwrap(); + assert!(fs.lookup_within(&alice, &alice, "work").await.is_ok()); + assert!(fs.lookup_within(&alice, &alice, "/work").await.is_ok()); + assert!(fs + .lookup_within(&alice, &alice, "work/../work") .await - .unwrap(); - let h = fs.open("k", OpenFlags::write_only()).await.unwrap(); - let target_mtime = std::time::UNIX_EPOCH + std::time::Duration::from_secs(1_700_000_000); - fs.set_times(&h, None, Some(target_mtime)).await.unwrap(); - // Wire-level: object metadata now carries our mtime key. - let head = backend.head_blob("k").await.unwrap(); - assert!(head - .metadata - .get(METADATA_MTIME_KEY) - .is_some_and(|v| v.starts_with("1700000000"))); + .is_ok()); } + /// An absolute path means the *scope's* root, not the filesystem's. This + /// is the first of the three escapes and the most obvious to try. #[tokio::test] - async fn set_times_at_with_no_follow_targets_symlink_itself() { - let (backend, fs) = small_fs(); - backend - .put_blob(PutBlobInput { - key: "target".into(), - body: Bytes::from_static(b"data"), - metadata: HashMap::new(), - content_type: None, - }) + async fn an_absolute_path_cannot_leave_the_scope() { + let (fs, alice, _) = tenanted().await; + assert!(fs + .lookup_within(&alice, &alice, "/tenants/bob") .await - .unwrap(); - fs.symlink_at(&fs.root(), "link", "target").await.unwrap(); - let mtime = std::time::UNIX_EPOCH + std::time::Duration::from_secs(1_700_000_000); - - // follow=false: writes mtime to the SYMLINK object, not the target. - fs.set_times_at(&fs.root(), "link", false, None, Some(mtime)) + .is_err()); + assert!(fs + .lookup_within(&alice, &alice, "/runtime-only") .await - .unwrap(); - let link_head = backend.head_blob("link").await.unwrap(); - assert!(link_head.metadata.contains_key(METADATA_MTIME_KEY)); - let target_head = backend.head_blob("target").await.unwrap(); - assert!(!target_head.metadata.contains_key(METADATA_MTIME_KEY)); - } - - // ---------- symlinks ---------- - - #[tokio::test] - async fn symlink_at_creates_object_with_metadata_flag() { - let (backend, fs) = small_fs(); + .is_err()); + // And it is not that the names are unknown — unscoped they resolve. let root = fs.root(); - let l = fs.symlink_at(&root, "link", "target.txt").await.unwrap(); - assert!(l.is_symlink()); - assert_eq!(l.symlink_target().as_deref(), Some("target.txt")); - // Verify the wire format. - let head = backend.head_blob("link").await.unwrap(); - assert_eq!( - head.metadata.get(SYMLINK_METADATA_KEY).map(|s| s.as_str()), - Some(SYMLINK_METADATA_VALUE) - ); - assert_eq!(head.content_type.as_deref(), Some(SYMLINK_CONTENT_TYPE)); - let g = backend.get_blob("link", None).await.unwrap(); - assert_eq!(&g.body[..], b"target.txt"); + assert!(fs.lookup_at(&root, "/tenants/bob").await.is_ok()); } + /// The second escape: no number of `..` walks out. #[tokio::test] - async fn symlink_at_eexist_via_conditional_put() { - let (_b, fs) = small_fs(); - let root = fs.root(); - fs.symlink_at(&root, "l", "a").await.unwrap(); - let r = fs.symlink_at(&root, "l", "b").await; - assert!(matches!(r, Err(FsError::AlreadyExists))); - } - - #[tokio::test] - async fn readlink_at_returns_target() { - let (_b, fs) = small_fs(); - let root = fs.root(); - fs.symlink_at(&root, "l", "deep/path").await.unwrap(); - assert_eq!(fs.readlink_at(&root, "l").await.unwrap(), "deep/path"); + async fn dot_dot_cannot_climb_out_of_the_scope() { + let (fs, alice, _) = tenanted().await; + for path in ["..", "../..", "../bob", "../../runtime-only", "../../.."] { + assert!( + fs.lookup_within(&alice, &alice, path).await.is_err() + || fs + .lookup_within(&alice, &alice, path) + .await + .map(|i| i.objid()) + .unwrap() + == alice.objid(), + "{path} escaped the scope" + ); + } } + /// The third, and the one a guest can build for itself: an absolute + /// symlink resolves from the scope too, so writing one buys nothing. #[tokio::test] - async fn readlink_at_on_regular_file_errors() { - let (backend, fs) = small_fs(); - backend - .put_blob(PutBlobInput { - key: "f".into(), - body: Bytes::from_static(b"x"), - metadata: HashMap::new(), - content_type: None, - }) + async fn an_absolute_symlink_cannot_leave_the_scope() { + let (fs, alice, _) = tenanted().await; + fs.symlink_at(&alice, "escape", "/tenants/bob/bobs-private") .await .unwrap(); - let r = fs.readlink_at(&fs.root(), "f").await; - assert!(matches!(r, Err(FsError::Invalid(_)))); - } - - // ---------- stat / read_dir ---------- + assert!(fs.lookup_within(&alice, &alice, "escape").await.is_err()); - #[tokio::test] - async fn stat_at_via_lookup_at() { - let (backend, fs) = small_fs(); - backend - .put_blob(PutBlobInput { - key: "a/b.txt".into(), - body: Bytes::from_static(b"hello"), - metadata: HashMap::new(), - content_type: None, - }) + fs.symlink_at(&alice, "up", "../../runtime-only") .await .unwrap(); - let attrs = fs.stat_at(&fs.root(), "a/b.txt").await.unwrap(); - assert_eq!(attrs.size, 5); + assert!(fs.lookup_within(&alice, &alice, "up").await.is_err()); } + /// `..` is not the only way to ask for a parent. A descriptor at the top + /// of a scope must not be walkable upwards either. #[tokio::test] - async fn parallel_upload_many_parts_produces_correct_content() { - // Stress the parallelization: pre-populate a 200-byte source (50 - // parts of 4 bytes each), then modify 17 non-contiguous parts and - // sync with max_parallel_parts=4. Verify byte-for-byte. - let backend = Arc::new(MemoryBackend::new()); - let cfg = Config::builder() - .part_schedule(crate::config::PartSchedule { - tiers: vec![(4, 100)], - }) - .single_part_threshold(4) // force MPU path - .max_parallel_parts(4) - .max_parallel_copy(4) - .max_merge_copy_bytes(1024) - .memory_limit_bytes(1024 * 1024) - .build(); - let fs = Fs::new(backend.clone() as Arc, Arc::new(cfg)); - - // Build source = "AAAA" repeated 50 times, but make each 4-byte - // chunk identifiable by a per-part letter so we can verify untouched - // parts later. - let mut source = Vec::with_capacity(200); - for i in 0..50 { - let c = b'a' + ((i % 26) as u8); - source.extend_from_slice(&[c; 4]); - } - backend - .put_blob(PutBlobInput { - key: "k".into(), - body: Bytes::from(source.clone()), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - - let h = fs - .open( - "k", - OpenFlags { - read: true, - write: true, - ..Default::default() - }, - ) - .await - .unwrap(); - - // Modify these part indices (non-contiguous, includes boundaries). - let modified_parts = [ - 0u32, 1, 5, 7, 11, 12, 13, 19, 23, 27, 31, 35, 41, 42, 43, 47, 49, - ]; - for &p in &modified_parts { - let off = (p * 4) as u64; - // Write distinctive 4-byte sequence "Z" packed; just use 'Z'. - fs.pwrite(&h, off, b"ZZZZ").await.unwrap(); - } - fs.sync(&h).await.unwrap(); - - let g = backend.get_blob("k", None).await.unwrap(); - assert_eq!(g.body.len(), 200, "size unchanged"); - for i in 0..200usize { - let part = (i / 4) as u32; - let expected = if modified_parts.contains(&part) { - b'Z' - } else { - b'a' + ((part % 26) as u8) - }; - assert_eq!(g.body[i], expected, "byte {i} (part {part})"); - } - assert_eq!(backend.mpu_count(), 0); - fs.close(&h).await.unwrap(); + async fn the_scope_is_its_own_parent() { + let (fs, alice, _) = tenanted().await; + let parent = fs.parent_within(&alice, &alice).await.unwrap(); + assert_eq!(parent.objid(), alice.objid()); + // Unscoped, the same call does reach `/tenants` — which is exactly the + // difference the scope makes. + assert_ne!(fs.parent_of(&alice).await.unwrap().objid(), alice.objid()); } + /// Two tenants, same relative path, different files. The separation is the + /// resolver's, not the guest's. #[tokio::test] - async fn read_dir_on_root_after_mkdir_and_write() { - let (_b, fs) = small_fs(); - fs.mkdir(&fs.root(), "d").await.unwrap(); - let h = fs - .open( - "f.txt", - OpenFlags { - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - fs.pwrite(&h, 0, b"x").await.unwrap(); - fs.sync(&h).await.unwrap(); - let snap = fs.read_dir(&fs.root()).await.unwrap(); - let names: Vec<_> = snap.iter().map(|e| e.name.as_str()).collect(); - assert!(names.contains(&"d")); - assert!(names.contains(&"f.txt")); - } - - // ---------- eager flusher (background upload) ---------- - - /// Write that fills more than `single_part_threshold` bytes across - /// multiple full parts kicks off an MPU mid-pwrite (begin happens - /// before pwrite returns). After sync, file contents round-trip. - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn pwrite_eager_starts_mpu_for_large_writes() { - let (backend, fs) = tiny_fs(); // 4 B/part, threshold = 8 B - let h = fs - .open( - "big.bin", - OpenFlags { - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - let payload = vec![b'A'; 16]; // 4 full parts, > threshold - let n = fs.pwrite(&h, 0, &payload).await.unwrap(); - assert_eq!(n, 16); - // Eager path begun an MPU. - assert_eq!(backend.mpu_count(), 1, "MPU should be begun by pwrite"); - // Sync completes the MPU; file content matches. - fs.sync(&h).await.unwrap(); - assert_eq!(backend.mpu_count(), 0, "MPU committed by sync"); - let g = backend.get_blob("big.bin", None).await.unwrap(); - assert_eq!(&g.body[..], &payload[..]); - } - - /// Sub-threshold writes never start an MPU: they go through the - /// single-PUT path on sync. - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn pwrite_below_threshold_does_not_eager_flush() { - let (backend, fs) = tiny_fs(); // threshold = 8 B - let h = fs - .open( - "small.bin", - OpenFlags { - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - fs.pwrite(&h, 0, b"abc").await.unwrap(); // 3 B, single part - assert_eq!(backend.mpu_count(), 0, "small write must not begin an MPU"); - fs.sync(&h).await.unwrap(); - let g = backend.get_blob("small.bin", None).await.unwrap(); - assert_eq!(&g.body[..], b"abc"); - } - - /// After an eager flush completes, the next sync sees no remaining - /// dirty parts for the eager-uploaded region — it just runs the MPU - /// commit without re-uploading. - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn sync_drains_inflight_then_commits() { - let (backend, fs) = tiny_fs(); - let h = fs - .open( - "drain.bin", - OpenFlags { - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - // Fill 3 full parts (12 B) — enqueues 3 eager UploadParts. - let payload = vec![b'X'; 12]; - fs.pwrite(&h, 0, &payload).await.unwrap(); - // Briefly yield to let the worker pick up the jobs. - tokio::task::yield_now().await; - // Sync drains in-flights, then commits. - fs.sync(&h).await.unwrap(); - let g = backend.get_blob("drain.bin", None).await.unwrap(); - assert_eq!(&g.body[..], &payload[..]); - } - - /// A second pwrite into the same part that has an in-flight eager - /// upload waits for the upload to land, then re-dirties the part. The - /// sync afterwards re-uploads the part with the new content. - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn second_pwrite_to_inflight_part_waits_then_redirties() { - let (backend, fs) = tiny_fs(); - let h = fs - .open( - "rewrite.bin", - OpenFlags { - write: true, - create: true, - ..Default::default() - }, - ) - .await - .unwrap(); - // First write fills 4 parts (16 B) — eager uploads triggered. - fs.pwrite(&h, 0, &[b'A'; 16]).await.unwrap(); - // Re-write the first part. absorb_inflight should await the eager - // upload then apply the new bytes; sync uploads the latest. - fs.pwrite(&h, 0, b"ZZZZ").await.unwrap(); - fs.sync(&h).await.unwrap(); - let g = backend.get_blob("rewrite.bin", None).await.unwrap(); - let mut expected = [b'A'; 16]; - expected[..4].copy_from_slice(b"ZZZZ"); - assert_eq!(&g.body[..], &expected[..]); + async fn two_scopes_name_different_files() { + let (fs, alice, bob) = tenanted().await; + fs.mkdir(&alice, "data").await.unwrap(); + fs.mkdir(&bob, "data").await.unwrap(); + let a = fs.lookup_within(&alice, &alice, "/data").await.unwrap(); + let b = fs.lookup_within(&bob, &bob, "/data").await.unwrap(); + assert_ne!(a.objid(), b.objid()); } } diff --git a/crates/s3fs-core/src/fs/tests.rs b/crates/s3fs-core/src/fs/tests.rs new file mode 100644 index 0000000..cb4da27 --- /dev/null +++ b/crates/s3fs-core/src/fs/tests.rs @@ -0,0 +1,1226 @@ +//! POSIX-semantics tests for [`Fs`]. +//! +//! The store layer is covered in `store::*`; these exercise the filesystem +//! behaviour built on top of it, and in particular the semantics that the +//! previous path-to-key engine could not provide. + +use super::*; +use crate::backend::memory::MemoryBackend; + +/// Two buckets that survive a remount, mirroring the deployment split. +struct Harness { + data: Arc, + roots: Arc, + config: Arc, +} + +impl Harness { + fn new() -> Self { + Harness { + data: Arc::new(MemoryBackend::new()), + roots: Arc::new(MemoryBackend::new()), + // Small records so multi-record files are cheap to write, and no + // retention so a test can inspect the buckets freely. + config: Arc::new( + Config::builder() + .record_size(4096) + .root_retention(None) + .build(), + ), + } + } + + /// Create on first use, mount thereafter. `Fs::mount` no longer formats an + /// empty store, so tests that want a filesystem have to say so. + async fn mount(&self) -> Arc { + let secret = MasterSecret::from_bytes([21u8; 32]); + match Fs::mount( + self.data.clone(), + self.roots.clone(), + &secret, + [9u8; 16], + self.config.clone(), + None, + ) + .await + { + Err(crate::FsError::NoFilesystem) => Fs::create( + self.data.clone(), + self.roots.clone(), + &secret, + [9u8; 16], + self.config.clone(), + ) + .await + .unwrap(), + other => other.unwrap(), + } + } +} + +async fn fs() -> Arc { + Harness::new().mount().await +} + +async fn write_file(fs: &Fs, path: &str, data: &[u8]) { + let h = fs.open(path, OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&h, 0, data).await.unwrap(); + fs.close(&h).await.unwrap(); +} + +async fn read_file(fs: &Fs, path: &str) -> Vec { + let h = fs.open(path, OpenFlags::read_only()).await.unwrap(); + let size = h.size().await; + let out = fs.pread(&h, 0, size as usize).await.unwrap(); + fs.close(&h).await.unwrap(); + out.to_vec() +} + +async fn names(fs: &Fs, dir: &Arc) -> Vec { + fs.read_dir(dir) + .await + .unwrap() + .into_iter() + .map(|e| e.name) + .collect() +} + +// ---- mount and namespace --------------------------------------------------- + +#[tokio::test] +async fn a_fresh_mount_has_an_empty_root_directory() { + let fs = fs().await; + let root = fs.root(); + assert_eq!(fs.stat(&root).await.unwrap().kind, InodeKind::Directory); + assert_eq!(names(&fs, &root).await, Vec::::new()); +} + +#[tokio::test] +async fn mkdir_then_list() { + let fs = fs().await; + let root = fs.root(); + fs.mkdir(&root, "b").await.unwrap(); + fs.mkdir(&root, "a").await.unwrap(); + + assert_eq!(names(&fs, &root).await, vec!["a", "b"]); + let a = fs.lookup_at(&root, "a").await.unwrap(); + assert_eq!(fs.stat(&a).await.unwrap().kind, InodeKind::Directory); +} + +#[tokio::test] +async fn mkdir_refuses_a_name_already_in_use() { + let fs = fs().await; + let root = fs.root(); + fs.mkdir(&root, "dup").await.unwrap(); + assert!(matches!( + fs.mkdir(&root, "dup").await, + Err(FsError::AlreadyExists) + )); +} + +#[tokio::test] +async fn nested_directories_resolve() { + let fs = fs().await; + let root = fs.root(); + let a = fs.mkdir(&root, "a").await.unwrap(); + let b = fs.mkdir(&a, "b").await.unwrap(); + fs.mkdir(&b, "c").await.unwrap(); + + let c = fs.lookup_at(&root, "a/b/c").await.unwrap(); + assert_eq!(fs.stat(&c).await.unwrap().kind, InodeKind::Directory); + assert!(matches!( + fs.lookup_at(&root, "a/nope/c").await, + Err(FsError::NotFound) + )); +} + +#[tokio::test] +async fn dotdot_walks_up_and_stops_at_the_root() { + let fs = fs().await; + let root = fs.root(); + let a = fs.mkdir(&root, "a").await.unwrap(); + fs.mkdir(&a, "b").await.unwrap(); + write_file(&fs, "/marker", b"x").await; + + let up = fs.lookup_at(&root, "a/b/../..").await.unwrap(); + assert_eq!(up.objid(), root.objid()); + + // `..` past the root must not escape the mount. + let clamped = fs.lookup_at(&root, "../../../marker").await.unwrap(); + assert_eq!(fs.stat(&clamped).await.unwrap().size, 1); +} + +#[tokio::test] +async fn dot_components_are_ignored() { + let fs = fs().await; + let root = fs.root(); + fs.mkdir(&root, "a").await.unwrap(); + write_file(&fs, "/a/f", b"hi").await; + assert_eq!(read_file(&fs, "/./a/./f").await, b"hi"); +} + +// ---- files ----------------------------------------------------------------- + +#[tokio::test] +async fn write_then_read_back() { + let fs = fs().await; + write_file(&fs, "/hello.txt", b"hello world").await; + assert_eq!(read_file(&fs, "/hello.txt").await, b"hello world"); +} + +#[tokio::test] +async fn a_writer_sees_its_own_unsynced_writes() { + let fs = fs().await; + let h = fs.open("/f", OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&h, 0, b"abcdef").await.unwrap(); + assert_eq!( + fs.pread(&h, 2, 3).await.unwrap(), + Bytes::from_static(b"cde") + ); + fs.close(&h).await.unwrap(); +} + +#[tokio::test] +async fn writes_spanning_many_records_round_trip() { + let fs = fs().await; + // 10 records at 4 KiB, written in one call. + let data: Vec = (0..40960u32).map(|i| (i % 251) as u8).collect(); + write_file(&fs, "/big", &data).await; + assert_eq!(read_file(&fs, "/big").await, data); +} + +#[tokio::test] +async fn a_write_in_the_middle_leaves_the_rest_intact() { + let fs = fs().await; + let data = vec![0xaau8; 12288]; + write_file(&fs, "/f", &data).await; + + let h = fs.open("/f", OpenFlags::read_write()).await.unwrap(); + fs.pwrite(&h, 5000, b"PATCH").await.unwrap(); + fs.close(&h).await.unwrap(); + + let mut expected = data.clone(); + expected[5000..5005].copy_from_slice(b"PATCH"); + assert_eq!(read_file(&fs, "/f").await, expected); +} + +#[tokio::test] +async fn sparse_writes_read_back_as_zeros() { + let fs = fs().await; + let h = fs.open("/sparse", OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&h, 100_000, b"end").await.unwrap(); + fs.close(&h).await.unwrap(); + + let out = read_file(&fs, "/sparse").await; + assert_eq!(out.len(), 100_003); + assert!(out[..100_000].iter().all(|&b| b == 0)); + assert_eq!(&out[100_000..], b"end"); +} + +#[tokio::test] +async fn reads_past_the_end_return_nothing() { + let fs = fs().await; + write_file(&fs, "/f", b"12345").await; + let h = fs.open("/f", OpenFlags::read_only()).await.unwrap(); + assert_eq!(fs.pread(&h, 5, 10).await.unwrap().len(), 0); + assert_eq!( + fs.pread(&h, 3, 10).await.unwrap(), + Bytes::from_static(b"45") + ); +} + +#[tokio::test] +async fn append_writes_at_the_end() { + let fs = fs().await; + write_file(&fs, "/log", b"one").await; + let h = fs + .open( + "/log", + OpenFlags { + write: true, + append: true, + ..Default::default() + }, + ) + .await + .unwrap(); + // The offset is ignored for an append handle. + fs.pwrite(&h, 0, b"-two").await.unwrap(); + fs.close(&h).await.unwrap(); + assert_eq!(read_file(&fs, "/log").await, b"one-two"); +} + +#[tokio::test] +async fn exclusive_create_refuses_an_existing_file() { + let fs = fs().await; + write_file(&fs, "/f", b"x").await; + assert!(matches!( + fs.open("/f", OpenFlags::create_new()).await, + Err(FsError::AlreadyExists) + )); +} + +#[tokio::test] +async fn opening_a_missing_file_without_create_fails() { + let fs = fs().await; + assert!(matches!( + fs.open("/nope", OpenFlags::read_only()).await, + Err(FsError::NotFound) + )); +} + +#[tokio::test] +async fn truncate_on_open_empties_the_file() { + let fs = fs().await; + write_file(&fs, "/f", b"content").await; + let h = fs + .open( + "/f", + OpenFlags { + read: true, + write: true, + truncate: true, + ..Default::default() + }, + ) + .await + .unwrap(); + assert_eq!(h.size().await, 0); + fs.close(&h).await.unwrap(); + assert_eq!(read_file(&fs, "/f").await, b""); +} + +#[tokio::test] +async fn set_size_truncates_and_grows() { + let fs = fs().await; + write_file(&fs, "/f", b"0123456789").await; + + let h = fs.open("/f", OpenFlags::read_write()).await.unwrap(); + fs.set_size(&h, 4).await.unwrap(); + fs.close(&h).await.unwrap(); + assert_eq!(read_file(&fs, "/f").await, b"0123"); + + // Growing is free: the new region is a hole, at any size. + let h = fs.open("/f", OpenFlags::read_write()).await.unwrap(); + fs.set_size(&h, 1_000_000).await.unwrap(); + fs.close(&h).await.unwrap(); + + let out = read_file(&fs, "/f").await; + assert_eq!(out.len(), 1_000_000); + assert_eq!(&out[..4], b"0123"); + assert!(out[4..].iter().all(|&b| b == 0)); +} + +#[tokio::test] +async fn a_regrown_file_does_not_resurrect_old_bytes() { + let fs = fs().await; + write_file(&fs, "/f", b"secretsecret").await; + let h = fs.open("/f", OpenFlags::read_write()).await.unwrap(); + fs.set_size(&h, 2).await.unwrap(); + fs.set_size(&h, 12).await.unwrap(); + fs.close(&h).await.unwrap(); + + let out = read_file(&fs, "/f").await; + assert_eq!(&out[..2], b"se"); + assert!(out[2..].iter().all(|&b| b == 0), "stale data reappeared"); +} + +// ---- durability ------------------------------------------------------------ + +#[tokio::test] +async fn committed_state_survives_a_remount() { + let h = Harness::new(); + let fs = h.mount().await; + fs.mkdir(&fs.root(), "dir").await.unwrap(); + write_file(&fs, "/dir/file", b"durable").await; + drop(fs); + + let fs = h.mount().await; + assert_eq!(read_file(&fs, "/dir/file").await, b"durable"); + assert_eq!(names(&fs, &fs.root()).await, vec!["dir"]); +} + +/// Dropping a handle without syncing loses the writes — but the filesystem is +/// still a consistent, verified state, never a torn one. +/// A sync that fails on a timeout keeps the buffered data, so the retry +/// (or the close) actually writes it. +#[tokio::test] +async fn a_failed_sync_keeps_the_writes_for_the_retry() { + let h = Harness::new(); + let fs = h.mount().await; + let handle = fs.open("/f", OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&handle, 0, b"survives").await.unwrap(); + + h.data.fail_next_put(crate::FsError::IoTimeout); + assert!(matches!( + fs.sync(&handle).await, + Err(crate::FsError::IoTimeout) + )); + fs.sync(&handle).await.unwrap(); + fs.close(&handle).await.unwrap(); + + let fs = h.mount().await; + assert_eq!(read_file(&fs, "/f").await, b"survives"); +} + +#[tokio::test] +async fn unsynced_writes_are_lost_but_leave_a_consistent_state() { + let h = Harness::new(); + let fs = h.mount().await; + write_file(&fs, "/f", b"committed").await; + + let handle = fs.open("/f", OpenFlags::read_write()).await.unwrap(); + fs.pwrite(&handle, 0, b"UNCOMMITTED").await.unwrap(); + drop(handle); // no close, no sync + drop(fs); + + let fs = h.mount().await; + assert_eq!(read_file(&fs, "/f").await, b"committed"); +} + +#[tokio::test] +async fn every_mutation_advances_the_root_sequence() { + let fs = fs().await; + let start = fs.store().root().await.unwrap().seq; + + fs.mkdir(&fs.root(), "a").await.unwrap(); + write_file(&fs, "/a/f", b"x").await; + + let end = fs.store().root().await.unwrap().seq; + assert!(end > start, "mutations must produce new anchored roots"); +} + +// ---- rename ---------------------------------------------------------------- + +#[tokio::test] +async fn rename_within_a_directory() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/old", b"data").await; + + fs.rename(&root, "old", &root, "new").await.unwrap(); + assert_eq!(names(&fs, &root).await, vec!["new"]); + assert_eq!(read_file(&fs, "/new").await, b"data"); + assert!(matches!( + fs.lookup_at(&root, "old").await, + Err(FsError::NotFound) + )); +} + +#[tokio::test] +async fn rename_across_directories() { + let fs = fs().await; + let root = fs.root(); + let a = fs.mkdir(&root, "a").await.unwrap(); + let b = fs.mkdir(&root, "b").await.unwrap(); + write_file(&fs, "/a/f", b"moved").await; + + fs.rename(&a, "f", &b, "g").await.unwrap(); + assert_eq!(names(&fs, &a).await, Vec::::new()); + assert_eq!(names(&fs, &b).await, vec!["g"]); + assert_eq!(read_file(&fs, "/b/g").await, b"moved"); +} + +/// The property the previous engine could not offer: a rename is one commit, +/// so no observer ever sees the entry under both names or under neither. +#[tokio::test] +async fn rename_is_a_single_commit() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/old", b"data").await; + + let before = fs.store().root().await.unwrap().seq; + fs.rename(&root, "old", &root, "new").await.unwrap(); + let after = fs.store().root().await.unwrap().seq; + assert_eq!(after, before + 1, "rename must be exactly one commit"); +} + +#[tokio::test] +async fn renaming_a_directory_moves_its_whole_subtree() { + let fs = fs().await; + let root = fs.root(); + let a = fs.mkdir(&root, "a").await.unwrap(); + fs.mkdir(&a, "sub").await.unwrap(); + write_file(&fs, "/a/sub/deep", b"deep").await; + + // A directory rename is one entry move, not a recursive copy. + let before = fs.store().root().await.unwrap().seq; + fs.rename(&root, "a", &root, "z").await.unwrap(); + assert_eq!(fs.store().root().await.unwrap().seq, before + 1); + + assert_eq!(read_file(&fs, "/z/sub/deep").await, b"deep"); +} + +/// The subtree would keep its parent link into a directory now inside it, +/// and nothing from the root could reach any of it again. +#[tokio::test] +async fn a_directory_cannot_be_renamed_into_its_own_subtree() { + let fs = fs().await; + let root = fs.root(); + let a = fs.mkdir(&root, "a").await.unwrap(); + let b = fs.mkdir(&a, "b").await.unwrap(); + write_file(&fs, "/a/b/data", b"data").await; + + assert!(matches!( + fs.rename(&root, "a", &b, "moved").await, + Err(FsError::Invalid(_)) + )); + assert!(matches!( + fs.rename(&root, "a", &a, "moved").await, + Err(FsError::Invalid(_)) + )); + assert_eq!(read_file(&fs, "/a/b/data").await, b"data"); + + // A file may go anywhere, and a directory may still go sideways. + fs.rename(&b, "data", &b, "data2").await.unwrap(); + fs.rename(&a, "b", &root, "b").await.unwrap(); + assert_eq!(read_file(&fs, "/b/data2").await, b"data"); +} + +#[tokio::test] +async fn renaming_a_directory_repoints_its_parent_link() { + let fs = fs().await; + let root = fs.root(); + let a = fs.mkdir(&root, "a").await.unwrap(); + let b = fs.mkdir(&root, "b").await.unwrap(); + let moved = fs.mkdir(&a, "moved").await.unwrap(); + + fs.rename(&a, "moved", &b, "moved").await.unwrap(); + + // `..` from inside the moved directory must lead to its new parent. + let up = fs.lookup_at(&moved, "..").await.unwrap(); + assert_eq!(up.objid(), b.objid()); +} + +#[tokio::test] +async fn rename_replaces_an_existing_file() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/src", b"new").await; + write_file(&fs, "/dst", b"old").await; + + fs.rename(&root, "src", &root, "dst").await.unwrap(); + assert_eq!(names(&fs, &root).await, vec!["dst"]); + assert_eq!(read_file(&fs, "/dst").await, b"new"); +} + +#[tokio::test] +async fn rename_refuses_mismatched_kinds_and_non_empty_targets() { + let fs = fs().await; + let root = fs.root(); + fs.mkdir(&root, "dir").await.unwrap(); + fs.mkdir(&root, "full").await.unwrap(); + write_file(&fs, "/full/inside", b"x").await; + write_file(&fs, "/file", b"x").await; + + assert!(matches!( + fs.rename(&root, "file", &root, "dir").await, + Err(FsError::IsDirectory) + )); + assert!(matches!( + fs.rename(&root, "dir", &root, "file").await, + Err(FsError::NotDirectory) + )); + assert!(matches!( + fs.rename(&root, "dir", &root, "full").await, + Err(FsError::NotEmpty) + )); +} + +#[tokio::test] +async fn renaming_onto_itself_is_a_no_op() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/f", b"x").await; + fs.rename(&root, "f", &root, "f").await.unwrap(); + assert_eq!(read_file(&fs, "/f").await, b"x"); +} + +// ---- unlink and rmdir ------------------------------------------------------ + +#[tokio::test] +async fn unlink_removes_a_file() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/f", b"x").await; + fs.unlink(&root, "f").await.unwrap(); + + assert_eq!(names(&fs, &root).await, Vec::::new()); + assert!(matches!( + fs.lookup_at(&root, "f").await, + Err(FsError::NotFound) + )); + assert!(matches!( + fs.unlink(&root, "f").await, + Err(FsError::NotFound) + )); +} + +#[tokio::test] +async fn rmdir_requires_an_empty_directory() { + let fs = fs().await; + let root = fs.root(); + fs.mkdir(&root, "d").await.unwrap(); + write_file(&fs, "/d/f", b"x").await; + + assert!(matches!(fs.rmdir(&root, "d").await, Err(FsError::NotEmpty))); + fs.unlink(&fs.lookup_at(&root, "d").await.unwrap(), "f") + .await + .unwrap(); + fs.rmdir(&root, "d").await.unwrap(); + assert_eq!(names(&fs, &root).await, Vec::::new()); +} + +#[tokio::test] +async fn unlink_and_rmdir_refuse_the_wrong_kind() { + let fs = fs().await; + let root = fs.root(); + fs.mkdir(&root, "d").await.unwrap(); + write_file(&fs, "/f", b"x").await; + + assert!(matches!( + fs.unlink(&root, "d").await, + Err(FsError::IsDirectory) + )); + assert!(matches!( + fs.rmdir(&root, "f").await, + Err(FsError::NotDirectory) + )); +} + +// ---- symlinks -------------------------------------------------------------- + +#[tokio::test] +async fn symlink_round_trips_and_resolves() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/target", b"pointed-at").await; + fs.symlink_at(&root, "link", "target").await.unwrap(); + + assert_eq!(fs.readlink_at(&root, "link").await.unwrap(), "target"); + assert_eq!(read_file(&fs, "/link").await, b"pointed-at"); + + // Without following, the link itself is what resolves. + let raw = fs.lookup_at_no_follow(&root, "link").await.unwrap(); + assert_eq!(fs.stat(&raw).await.unwrap().kind, InodeKind::Symlink); +} + +#[tokio::test] +async fn symlinks_resolve_through_directories() { + let fs = fs().await; + let root = fs.root(); + let a = fs.mkdir(&root, "a").await.unwrap(); + fs.mkdir(&a, "real").await.unwrap(); + write_file(&fs, "/a/real/f", b"deep").await; + fs.symlink_at(&root, "shortcut", "a/real").await.unwrap(); + + assert_eq!(read_file(&fs, "/shortcut/f").await, b"deep"); +} + +#[tokio::test] +async fn an_absolute_symlink_resolves_from_the_root() { + let fs = fs().await; + let root = fs.root(); + fs.mkdir(&root, "d").await.unwrap(); + write_file(&fs, "/d/f", b"abs").await; + let d = fs.lookup_at(&root, "d").await.unwrap(); + fs.symlink_at(&d, "link", "/d/f").await.unwrap(); + + assert_eq!(read_file(&fs, "/d/link").await, b"abs"); +} + +#[tokio::test] +async fn a_symlink_loop_is_bounded() { + let fs = fs().await; + let root = fs.root(); + fs.symlink_at(&root, "a", "b").await.unwrap(); + fs.symlink_at(&root, "b", "a").await.unwrap(); + + assert!(matches!(fs.lookup_at(&root, "a").await, Err(FsError::Loop))); +} + +#[tokio::test] +async fn a_long_symlink_target_spills_to_blocks_and_reads_back() { + let fs = fs().await; + let root = fs.root(); + // Longer than the dnode's inline capacity, so it must use data blocks. + let target = format!("/{}", "x".repeat(INLINE_CAP + 50)); + fs.symlink_at(&root, "long", &target).await.unwrap(); + assert_eq!(fs.readlink_at(&root, "long").await.unwrap(), target); +} + +#[tokio::test] +async fn readlink_refuses_a_regular_file() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/f", b"x").await; + assert!(fs.readlink_at(&root, "f").await.is_err()); +} + +// ---- attributes ------------------------------------------------------------ + +#[tokio::test] +async fn stat_reports_real_metadata() { + let fs = fs().await; + write_file(&fs, "/f", b"12345").await; + let ino = fs.lookup_at(&fs.root(), "f").await.unwrap(); + let a = fs.stat(&ino).await.unwrap(); + + assert_eq!(a.kind, InodeKind::RegularFile); + assert_eq!(a.size, 5); + assert_eq!(a.nlink, 1); + assert_eq!(a.mode, DEFAULT_FILE_MODE); +} + +/// Timestamps live in the dnode, so there is no S3 `LastModified` to disagree +/// with and no self-copy to persist them. +#[tokio::test] +async fn set_times_persists_exactly() { + let fs = fs().await; + write_file(&fs, "/f", b"x").await; + + let when = crate::inode::from_nanos(1_700_000_000_000_000_000); + let h = fs.open("/f", OpenFlags::read_write()).await.unwrap(); + fs.set_times(&h, Some(when), Some(when)).await.unwrap(); + fs.close(&h).await.unwrap(); + + let ino = fs.lookup_at(&fs.root(), "f").await.unwrap(); + let a = fs.stat(&ino).await.unwrap(); + assert_eq!(a.mtime, when); + assert_eq!(a.atime, when); +} + +#[tokio::test] +async fn set_times_at_works_without_a_handle() { + let fs = fs().await; + let root = fs.root(); + fs.mkdir(&root, "d").await.unwrap(); + + let when = crate::inode::from_nanos(1_234_567_890); + fs.set_times_at(&root, "d", None, Some(when), true) + .await + .unwrap(); + + let d = fs.lookup_at(&root, "d").await.unwrap(); + assert_eq!(fs.stat(&d).await.unwrap().mtime, when); +} + +/// Object ids are never reused, so identity holds across a remount — which is +/// what `is-same-object` and `metadata-hash` need in order to be exact. +#[tokio::test] +async fn identity_is_stable_across_a_remount() { + let h = Harness::new(); + let fs = h.mount().await; + write_file(&fs, "/f", b"x").await; + let before = *fs.lookup_at(&fs.root(), "f").await.unwrap(); + drop(fs); + + let fs = h.mount().await; + let after = *fs.lookup_at(&fs.root(), "f").await.unwrap(); + assert_eq!(before, after); +} + +// ---- validation ------------------------------------------------------------ + +#[tokio::test] +async fn bad_names_are_rejected() { + let fs = fs().await; + let root = fs.root(); + assert!(fs.mkdir(&root, "with\0nul").await.is_err()); + assert!(fs.mkdir(&root, &"x".repeat(300)).await.is_err()); + assert!(fs.mkdir(&root, "").await.is_err()); +} + +#[tokio::test] +async fn opening_with_neither_read_nor_write_is_rejected() { + let fs = fs().await; + assert!(matches!( + fs.open("/f", OpenFlags::default()).await, + Err(FsError::Invalid(_)) + )); +} + +/// Every mutator that takes a handle, not just the one that writes bytes. +/// `set_times` used to be exempt, which made `flags.write` mean "may change the +/// contents" rather than "may change the file". +#[tokio::test] +async fn mutating_through_a_read_only_handle_is_refused() { + let fs = fs().await; + write_file(&fs, "/f", b"x").await; + let h = fs.open("/f", OpenFlags::read_only()).await.unwrap(); + + assert!(matches!( + fs.pwrite(&h, 0, b"nope").await, + Err(FsError::BadDescriptor) + )); + assert!(matches!( + fs.set_size(&h, 0).await, + Err(FsError::BadDescriptor) + )); + let when = crate::inode::from_nanos(1_700_000_000_000_000_000); + assert!(matches!( + fs.set_times(&h, Some(when), Some(when)).await, + Err(FsError::BadDescriptor) + )); +} + +/// Why the gate matters here more than it would in an ordinary filesystem: +/// every mutation commits, and every commit publishes a signed root record +/// under Object Lock that nothing can afterwards remove. A refusal that still +/// advanced the chain would be worse than no refusal, because it would look +/// safe. +#[tokio::test] +async fn a_read_only_handle_cannot_advance_the_anchor_chain() { + let fs = fs().await; + write_file(&fs, "/f", b"x").await; + let before = fs.store().root().await.unwrap().seq; + + let h = fs.open("/f", OpenFlags::read_only()).await.unwrap(); + let when = crate::inode::from_nanos(1_700_000_000_000_000_000); + let _ = fs.pwrite(&h, 0, b"nope").await; + let _ = fs.set_size(&h, 0).await; + let _ = fs.set_times(&h, Some(when), Some(when)).await; + fs.close(&h).await.unwrap(); + + assert_eq!( + fs.store().root().await.unwrap().seq, + before, + "a read-only handle committed something" + ); + + // And the timestamps it tried to set did not land. + let ino = fs.lookup_at(&fs.root(), "f").await.unwrap(); + assert_ne!(fs.stat(&ino).await.unwrap().mtime, when); +} + +#[tokio::test] +async fn opening_a_directory_for_writing_is_refused() { + let fs = fs().await; + fs.mkdir(&fs.root(), "d").await.unwrap(); + assert!(matches!( + fs.open("/d", OpenFlags::write_only()).await, + Err(FsError::IsDirectory) + )); +} + +#[tokio::test] +async fn resolving_through_a_file_is_refused() { + let fs = fs().await; + write_file(&fs, "/f", b"x").await; + assert!(matches!( + fs.lookup_at(&fs.root(), "f/inside").await, + Err(FsError::NotDirectory) + )); +} + +// ---- integrity end to end -------------------------------------------------- + +/// The whole point of the design, exercised through the public API: corrupt a +/// slab and the filesystem refuses to serve the bytes rather than returning +/// plausible-looking wrong ones. +#[tokio::test] +async fn tampered_storage_is_refused_at_the_filesystem_layer() { + use crate::backend::PutBlobInput; + + let h = Harness::new(); + let fs = h.mount().await; + write_file(&fs, "/f", &vec![0x5au8; 8192]).await; + drop(fs); + + // Flip a byte in every slab the data bucket holds. + for key in h.data.keys() { + let mut body = h.data.get_blob(&key, None).await.unwrap().body.to_vec(); + body[0] ^= 0xff; + h.data + .put_blob(PutBlobInput::new(key, Bytes::from(body))) + .await + .unwrap(); + } + + let fs = h.mount().await; + let result = async { + let handle = fs.open("/f", OpenFlags::read_only()).await?; + fs.pread(&handle, 0, 8192).await + } + .await; + assert!( + matches!(result, Err(FsError::Integrity(_))), + "expected an integrity failure, got {result:?}" + ); +} + +// ---- hard links ------------------------------------------------------------ + +#[tokio::test] +async fn a_hard_link_shares_one_object() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/original", b"shared").await; + + fs.link_at(&root, "original", &root, "alias", true) + .await + .unwrap(); + + assert_eq!(names(&fs, &root).await, vec!["alias", "original"]); + assert_eq!(read_file(&fs, "/alias").await, b"shared"); + + // Both names must be the same object, not a copy of it. + let a = fs.lookup_at(&root, "original").await.unwrap(); + let b = fs.lookup_at(&root, "alias").await.unwrap(); + assert_eq!(*a, *b); + assert_eq!(fs.stat(&a).await.unwrap().nlink, 2); +} + +#[tokio::test] +async fn a_write_through_one_link_is_visible_through_the_other() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/a", b"before").await; + fs.link_at(&root, "a", &root, "b", true).await.unwrap(); + + let h = fs.open("/b", OpenFlags::read_write()).await.unwrap(); + fs.pwrite(&h, 0, b"after!").await.unwrap(); + fs.close(&h).await.unwrap(); + + assert_eq!(read_file(&fs, "/a").await, b"after!"); +} + +#[tokio::test] +async fn unlinking_one_link_leaves_the_other() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/a", b"content").await; + fs.link_at(&root, "a", &root, "b", true).await.unwrap(); + + fs.unlink(&root, "a").await.unwrap(); + assert_eq!(read_file(&fs, "/b").await, b"content"); + + let b = fs.lookup_at(&root, "b").await.unwrap(); + assert_eq!(fs.stat(&b).await.unwrap().nlink, 1); + assert!(matches!( + fs.lookup_at(&root, "a").await, + Err(FsError::NotFound) + )); + + // Removing the last link really does remove the object. + fs.unlink(&root, "b").await.unwrap(); + assert_eq!(names(&fs, &root).await, Vec::::new()); +} + +#[tokio::test] +async fn hard_links_survive_a_remount() { + let h = Harness::new(); + let fs = h.mount().await; + let root = fs.root(); + write_file(&fs, "/a", b"durable").await; + fs.link_at(&root, "a", &root, "b", true).await.unwrap(); + drop(fs); + + let fs = h.mount().await; + let a = fs.lookup_at(&fs.root(), "a").await.unwrap(); + let b = fs.lookup_at(&fs.root(), "b").await.unwrap(); + assert_eq!(*a, *b); + assert_eq!(fs.stat(&a).await.unwrap().nlink, 2); +} + +#[tokio::test] +async fn links_can_cross_directories() { + let fs = fs().await; + let root = fs.root(); + let d = fs.mkdir(&root, "d").await.unwrap(); + write_file(&fs, "/f", b"x").await; + + fs.link_at(&root, "f", &d, "linked", true).await.unwrap(); + assert_eq!(read_file(&fs, "/d/linked").await, b"x"); +} + +/// A hard link to a directory would turn the namespace into a graph, and the +/// dnode's parent link has room for exactly one answer. +#[tokio::test] +async fn a_directory_cannot_be_hard_linked() { + let fs = fs().await; + let root = fs.root(); + fs.mkdir(&root, "d").await.unwrap(); + assert!(matches!( + fs.link_at(&root, "d", &root, "alias", true).await, + Err(FsError::NotPermitted) + )); +} + +#[tokio::test] +async fn link_refuses_an_occupied_name_or_a_missing_source() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/a", b"x").await; + write_file(&fs, "/b", b"y").await; + + assert!(matches!( + fs.link_at(&root, "a", &root, "b", true).await, + Err(FsError::AlreadyExists) + )); + assert!(matches!( + fs.link_at(&root, "nope", &root, "c", true).await, + Err(FsError::NotFound) + )); +} + +#[tokio::test] +async fn linking_a_symlink_can_target_the_link_itself() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/target", b"data").await; + fs.symlink_at(&root, "sym", "target").await.unwrap(); + + // Without following, the new name is a second link to the symlink object. + fs.link_at(&root, "sym", &root, "sym2", false) + .await + .unwrap(); + assert_eq!(fs.readlink_at(&root, "sym2").await.unwrap(), "target"); + + // Following, it is a second link to the file the symlink points at. + fs.link_at(&root, "sym", &root, "direct", true) + .await + .unwrap(); + let direct = fs.lookup_at_no_follow(&root, "direct").await.unwrap(); + assert_eq!(fs.stat(&direct).await.unwrap().kind, InodeKind::RegularFile); +} + +// ---- unlink while open ----------------------------------------------------- + +/// POSIX keeps an unlinked file alive until the last descriptor closes. The +/// previous engine could not do this at all. +#[tokio::test] +async fn an_unlinked_file_stays_readable_through_an_open_handle() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/doomed", b"still here").await; + + let h = fs.open("/doomed", OpenFlags::read_only()).await.unwrap(); + fs.unlink(&root, "doomed").await.unwrap(); + + // The name is gone... + assert!(matches!( + fs.lookup_at(&root, "doomed").await, + Err(FsError::NotFound) + )); + // ...but the handle still works. + assert_eq!( + fs.pread(&h, 0, 10).await.unwrap(), + Bytes::from_static(b"still here") + ); + fs.close(&h).await.unwrap(); +} + +#[tokio::test] +async fn an_unlinked_file_is_still_writable_through_its_handle() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/doomed", b"old").await; + + let h = fs.open("/doomed", OpenFlags::read_write()).await.unwrap(); + fs.unlink(&root, "doomed").await.unwrap(); + fs.pwrite(&h, 0, b"new").await.unwrap(); + assert_eq!( + fs.pread(&h, 0, 3).await.unwrap(), + Bytes::from_static(b"new") + ); + fs.close(&h).await.unwrap(); +} + +#[tokio::test] +async fn the_object_is_freed_when_the_last_handle_closes() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/doomed", b"x").await; + let ino = fs.lookup_at(&root, "doomed").await.unwrap(); + + let h = fs.open("/doomed", OpenFlags::read_only()).await.unwrap(); + fs.unlink(&root, "doomed").await.unwrap(); + assert!(fs.stat(&ino).await.is_ok(), "still alive while open"); + + fs.close(&h).await.unwrap(); + assert!( + matches!(fs.stat(&ino).await, Err(FsError::NotFound)), + "the last close must free it" + ); +} + +#[tokio::test] +async fn the_object_survives_until_every_handle_closes() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/doomed", b"x").await; + let ino = fs.lookup_at(&root, "doomed").await.unwrap(); + + let a = fs.open("/doomed", OpenFlags::read_only()).await.unwrap(); + let b = fs.open("/doomed", OpenFlags::read_only()).await.unwrap(); + fs.unlink(&root, "doomed").await.unwrap(); + + fs.close(&a).await.unwrap(); + assert!(fs.stat(&ino).await.is_ok(), "one handle remains"); + fs.close(&b).await.unwrap(); + assert!(matches!(fs.stat(&ino).await, Err(FsError::NotFound))); +} + +/// A closed file with no remaining handles is freed immediately, not deferred. +#[tokio::test] +async fn unlinking_a_closed_file_frees_it_at_once() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/f", b"x").await; + let ino = fs.lookup_at(&root, "f").await.unwrap(); + + fs.unlink(&root, "f").await.unwrap(); + assert!(matches!(fs.stat(&ino).await, Err(FsError::NotFound))); +} + +/// If the object gets a new name before the last close, it must not be freed. +#[tokio::test] +async fn relinking_before_the_last_close_cancels_the_free() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/f", b"rescued").await; + + let h = fs.open("/f", OpenFlags::read_only()).await.unwrap(); + fs.link_at(&root, "f", &root, "rescue", true).await.unwrap(); + fs.unlink(&root, "f").await.unwrap(); + fs.close(&h).await.unwrap(); + + assert_eq!(read_file(&fs, "/rescue").await, b"rescued"); +} + +#[tokio::test] +async fn a_file_replaced_by_rename_stays_readable_while_open() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/victim", b"victim data").await; + write_file(&fs, "/src", b"src data").await; + + let h = fs.open("/victim", OpenFlags::read_only()).await.unwrap(); + fs.rename(&root, "src", &root, "victim").await.unwrap(); + + assert_eq!(read_file(&fs, "/victim").await, b"src data"); + assert_eq!( + fs.pread(&h, 0, 11).await.unwrap(), + Bytes::from_static(b"victim data"), + "the replaced file must remain readable through its handle" + ); + fs.close(&h).await.unwrap(); +} + +// ---- snapshots ------------------------------------------------------------- + +/// Every root record is a snapshot, because copy-on-write never overwrites the +/// blocks an older root names. +#[tokio::test] +async fn a_snapshot_shows_the_state_at_its_sequence() { + let fs = fs().await; + write_file(&fs, "/f", b"version one").await; + let first = fs.store().root().await.unwrap().seq; + + let h = fs.open("/f", OpenFlags::read_write()).await.unwrap(); + fs.pwrite(&h, 0, b"version two").await.unwrap(); + fs.close(&h).await.unwrap(); + + assert_eq!(read_file(&fs, "/f").await, b"version two"); + + let snap = fs.open_snapshot(first).await.unwrap(); + let ino = snap.lookup(&snap.root(), "f").await.unwrap(); + assert_eq!( + snap.read(&ino, 0, 11).await.unwrap(), + Bytes::from_static(b"version one") + ); + assert_eq!(snap.info().seq, first); +} + +#[tokio::test] +async fn a_snapshot_still_holds_a_since_deleted_file() { + let fs = fs().await; + let root = fs.root(); + write_file(&fs, "/gone", b"recoverable").await; + let before_delete = fs.store().root().await.unwrap().seq; + fs.unlink(&root, "gone").await.unwrap(); + + assert!(matches!( + fs.lookup_at(&root, "gone").await, + Err(FsError::NotFound) + )); + + let snap = fs.open_snapshot(before_delete).await.unwrap(); + let ino = snap.lookup(&snap.root(), "gone").await.unwrap(); + assert_eq!( + snap.read(&ino, 0, 11).await.unwrap(), + Bytes::from_static(b"recoverable") + ); +} + +#[tokio::test] +async fn snapshots_list_newest_first_with_distinct_merkle_roots() { + let fs = fs().await; + for i in 0..4u32 { + write_file(&fs, &format!("/f{i}"), b"x").await; + } + + let snaps = fs.snapshots(3).await.unwrap(); + assert_eq!(snaps.len(), 3); + assert!( + snaps.windows(2).all(|w| w[0].seq > w[1].seq), + "newest first" + ); + // Each commit changes the filesystem, so each root covers a different tree. + let roots: std::collections::HashSet<_> = snaps.iter().map(|s| s.merkle_root).collect(); + assert_eq!(roots.len(), snaps.len()); +} + +#[tokio::test] +async fn snapshot_listing_stops_at_the_genesis_root() { + let fs = fs().await; + write_file(&fs, "/f", b"x").await; + + // Asking for more than exist yields what exists, ending at sequence 0. + let snaps = fs.snapshots(1000).await.unwrap(); + assert_eq!(snaps.last().unwrap().seq, 0); + assert!(fs.snapshots(0).await.unwrap().is_empty()); +} + +#[tokio::test] +async fn a_snapshot_can_list_directories_and_stat() { + let fs = fs().await; + let root = fs.root(); + fs.mkdir(&root, "d").await.unwrap(); + write_file(&fs, "/d/a", b"12345").await; + let seq = fs.store().root().await.unwrap().seq; + fs.unlink(&fs.lookup_at(&root, "d").await.unwrap(), "a") + .await + .unwrap(); + + let snap = fs.open_snapshot(seq).await.unwrap(); + let d = snap.lookup(&snap.root(), "d").await.unwrap(); + let entries: Vec<_> = snap + .read_dir(&d) + .await + .unwrap() + .into_iter() + .map(|e| e.name) + .collect(); + assert_eq!(entries, vec!["a"]); + + let a = snap.lookup(&snap.root(), "d/a").await.unwrap(); + assert_eq!(snap.stat(&a).await.unwrap().size, 5); +} + +/// Opening a snapshot must not disturb the live mount's rollback floor — +/// otherwise reading history would be a way to make the store rewind. +#[tokio::test] +async fn opening_a_snapshot_does_not_lower_the_rollback_floor() { + let fs = fs().await; + write_file(&fs, "/a", b"x").await; + write_file(&fs, "/b", b"x").await; + + let floor = fs.store().roots().expected_seq(); + let _snap = fs.open_snapshot(0).await.unwrap(); + assert_eq!(fs.store().roots().expected_seq(), floor); + + // And the live mount is untouched. + assert_eq!(read_file(&fs, "/b").await, b"x"); +} diff --git a/crates/s3fs-core/src/inode.rs b/crates/s3fs-core/src/inode.rs new file mode 100644 index 0000000..dc10092 --- /dev/null +++ b/crates/s3fs-core/src/inode.rs @@ -0,0 +1,199 @@ +//! `Inode` — a handle on one object, and the attributes callers see. +//! +//! An inode is just an object id. There is no cached metadata and no parent +//! pointer: the dnode in the object set is the only source of truth, and it is +//! cheap to read because its block sits in the decrypted-block cache. That is +//! a deliberate change from the previous engine, whose inode tree cached +//! attributes with a TTL that was never actually checked — so a cache could +//! disagree with the store indefinitely and nothing would notice. +//! +//! Identity is `(objid, gen)`. Object ids are never reused, so this is stable +//! across mounts, which is what makes `is-same-object` and `metadata-hash` +//! exact rather than a hash of an ETag. + +use std::sync::Arc; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use crate::errors::{FsError, FsResult}; +use crate::store::dnode::{Dnode, DnodeKind}; + +/// Stable identifier for an object. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)] +pub struct InodeId(pub u64); + +impl InodeId { + pub fn get(self) -> u64 { + self.0 + } +} + +/// What an object is, as the WASI layer sees it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum InodeKind { + RegularFile, + Directory, + Symlink, +} + +impl InodeKind { + pub(crate) fn from_dnode(kind: DnodeKind) -> FsResult { + Ok(match kind { + DnodeKind::File => InodeKind::RegularFile, + DnodeKind::Dir => InodeKind::Directory, + DnodeKind::Symlink => InodeKind::Symlink, + DnodeKind::Free => return Err(FsError::NotFound), + DnodeKind::DnodeArray => return Err(FsError::NotSupported), + }) + } + + /// Used by callers that create objects of a chosen kind. + pub fn to_dnode(self) -> DnodeKind { + match self { + InodeKind::RegularFile => DnodeKind::File, + InodeKind::Directory => DnodeKind::Dir, + InodeKind::Symlink => DnodeKind::Symlink, + } + } + + pub fn is_dir(self) -> bool { + matches!(self, InodeKind::Directory) + } +} + +/// A reference to one object. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub struct Inode { + pub id: InodeId, + /// Generation, paired with the id to form an identity that survives a + /// remount. + pub gen: u64, +} + +impl Inode { + pub fn new(objid: u64, gen: u64) -> Arc { + Arc::new(Inode { + id: InodeId(objid), + gen, + }) + } + + pub fn objid(&self) -> u64 { + self.id.0 + } +} + +/// Everything `stat` reports. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Attrs { + pub kind: InodeKind, + /// Logical size in bytes. For a directory this is the size of its block + /// structure, not an entry count. + pub size: u64, + /// Number of directory entries pointing at this object. + pub nlink: u32, + pub mode: u32, + pub atime: SystemTime, + pub mtime: SystemTime, + pub ctime: SystemTime, + pub btime: SystemTime, +} + +impl Attrs { + pub(crate) fn from_dnode(d: &Dnode) -> FsResult { + Ok(Attrs { + kind: InodeKind::from_dnode(d.kind)?, + size: d.size, + nlink: d.nlink, + mode: d.mode, + atime: from_nanos(d.atime_nanos), + mtime: from_nanos(d.mtime_nanos), + ctime: from_nanos(d.ctime_nanos), + btime: from_nanos(d.btime_nanos), + }) + } +} + +/// One entry of a directory listing. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct DirEntry { + pub name: String, + pub kind: InodeKind, + /// Object id, so a caller can compare identity without a second lookup. + pub objid: u64, +} + +pub(crate) fn from_nanos(nanos: u64) -> SystemTime { + UNIX_EPOCH + Duration::from_nanos(nanos) +} + +pub(crate) fn to_nanos(t: SystemTime) -> u64 { + t.duration_since(UNIX_EPOCH) + .map(|d| d.as_nanos().min(u64::MAX as u128) as u64) + .unwrap_or(0) +} + +pub(crate) fn now_nanos() -> u64 { + to_nanos(SystemTime::now()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn kind_maps_both_ways() { + for k in [ + InodeKind::RegularFile, + InodeKind::Directory, + InodeKind::Symlink, + ] { + assert_eq!(InodeKind::from_dnode(k.to_dnode()).unwrap(), k); + } + } + + /// A free slot is "no such file", not a kind of file. Mapping it to + /// anything else would let a deleted object be stat'd. + #[test] + fn a_free_slot_is_not_a_file() { + assert!(matches!( + InodeKind::from_dnode(DnodeKind::Free), + Err(FsError::NotFound) + )); + assert!(InodeKind::from_dnode(DnodeKind::DnodeArray).is_err()); + } + + #[test] + fn timestamps_round_trip() { + for nanos in [0u64, 1, 1_000_000_000, 1_700_000_000_123_456_789] { + assert_eq!(to_nanos(from_nanos(nanos)), nanos); + } + } + + #[test] + fn pre_epoch_times_clamp_rather_than_panic() { + assert_eq!(to_nanos(UNIX_EPOCH - Duration::from_secs(1)), 0); + } + + #[test] + fn attrs_come_from_the_dnode() { + let mut d = Dnode::new(5, DnodeKind::File, 12, 42); + d.size = 1234; + d.mode = 0o100644; + d.nlink = 2; + let a = Attrs::from_dnode(&d).unwrap(); + + assert_eq!(a.kind, InodeKind::RegularFile); + assert_eq!(a.size, 1234); + assert_eq!(a.nlink, 2, "link count is real, not hardcoded to 1"); + assert_eq!(a.mode, 0o100644); + assert_eq!(to_nanos(a.mtime), 42); + } + + #[test] + fn identity_pairs_the_id_with_the_generation() { + let a = Inode::new(7, 1); + let b = Inode::new(7, 2); + assert_eq!(a.objid(), b.objid()); + assert_ne!(*a, *b, "a reused id with a new generation is a new object"); + } +} diff --git a/crates/s3fs-core/src/inode/attrs.rs b/crates/s3fs-core/src/inode/attrs.rs deleted file mode 100644 index 793995e..0000000 --- a/crates/s3fs-core/src/inode/attrs.rs +++ /dev/null @@ -1,275 +0,0 @@ -//! Inode data structures: ID, kind, attributes, and the in-memory tree node. -//! -//! Inodes are reference-counted via `Arc`. Parent references are `Weak` so a -//! drop of the tree's by-id table releases everything cleanly. Per-inode state -//! lives behind `parking_lot::RwLock`s that are short-held and never crossed -//! by `await` points. - -use std::collections::HashMap; -use std::num::NonZeroU64; -use std::sync::{Arc, Weak}; -use std::time::{Instant, SystemTime}; - -use parking_lot::RwLock; - -/// Stable per-`InodeTree` identifier. Root is always `InodeId(1)`. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)] -pub struct InodeId(pub NonZeroU64); - -impl InodeId { - /// The root of every `InodeTree`. - pub const ROOT: InodeId = InodeId(match NonZeroU64::new(1) { - Some(v) => v, - None => unreachable!(), - }); - - pub fn get(self) -> u64 { - self.0.get() - } -} - -impl std::fmt::Display for InodeId { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "ino:{}", self.0.get()) - } -} - -/// What this inode represents in the storage layer. -#[derive(Debug, Clone)] -pub enum InodeKind { - RegularFile, - /// Directory. `explicit_marker` records whether a zero-byte `dir/` - /// object exists in S3 (set on `mkdir`) or whether the directory is - /// implicit-from-prefix only. - Directory { - explicit_marker: bool, - }, - /// Symlink. `target` is the literal stored target path string. Resolution - /// happens at follow-time; at attr-cache level we just remember the body. - Symlink { - target: String, - }, -} - -impl InodeKind { - pub fn is_dir(&self) -> bool { - matches!(self, InodeKind::Directory { .. }) - } - pub fn is_symlink(&self) -> bool { - matches!(self, InodeKind::Symlink { .. }) - } - pub fn is_regular_file(&self) -> bool { - matches!(self, InodeKind::RegularFile) - } -} - -/// State machine for an inode's relationship to the backend. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum InodeState { - /// In sync with what we last saw from S3. - Cached, - /// Local writes pending; flusher will reconcile. (Reserved for the - /// upcoming buffer-pool / MPU layers.) - Modified, - /// Tombstoned by `unlink`/`rmdir`. Subsequent ops on stale handles must - /// return [`crate::errors::FsError::NotFound`]. - Deleted, -} - -/// Cached attribute snapshot. -#[derive(Debug, Clone)] -pub struct Attrs { - pub size: u64, - pub etag: String, - pub last_modified: SystemTime, - pub content_type: Option, - pub metadata: HashMap, - /// When this snapshot was fetched / last validated. Used for TTL checks - /// against `Config::attr_cache_ttl`. - pub fetched_at: Instant, -} - -impl Attrs { - /// Synthesize attrs for an implicit directory (no S3 object backs it). - pub fn synthetic_implicit_dir() -> Self { - Self { - size: 0, - etag: String::new(), - last_modified: SystemTime::UNIX_EPOCH, - content_type: None, - metadata: HashMap::new(), - fetched_at: Instant::now(), - } - } -} - -/// Mutable per-directory child table. -#[derive(Debug, Default)] -pub struct Children { - pub by_name: HashMap>, - /// `true` once a complete directory listing has been ingested; lookup - /// misses can then short-circuit to `NotFound` without hitting S3. - pub listing_complete: bool, - /// When the listing snapshot was taken. - pub listing_fetched_at: Option, -} - -/// Tracks an async rename that has rewired the inode tree but whose -/// background copy+delete in S3 has not yet completed. -/// -/// While set, `InodeTree::current_s3_key` returns `old_key` (so reads and -/// writes still resolve to the existing object). Once the worker finishes -/// the copy + delete, it clears this state and signals `done`. -/// -/// On worker error, `error` is populated and `state` is left set so the -/// next `Fs::sync` / `Fs::rename` for the inode can surface it. -#[derive(Debug)] -pub struct RenameState { - pub old_key: String, - /// `Some(_)` after a worker error. - pub error: RwLock>, - pub done: Arc, -} - -impl RenameState { - pub fn new(old_key: String) -> Arc { - Arc::new(Self { - old_key, - error: RwLock::new(None), - done: Arc::new(tokio::sync::Notify::new()), - }) - } -} - -/// In-memory inode. Cheap to clone (`Arc`). -#[derive(Debug)] -pub struct Inode { - pub id: InodeId, - /// Basename only. The root has `name == ""`. - pub name: String, - pub kind: RwLock, - pub attrs: RwLock, - pub state: RwLock, - pub parent: Option>, - pub children: RwLock, - /// Set on the *destination* inode of an async rename — the in-memory - /// rewire happened but the underlying S3 copy+delete is still in - /// flight. Cleared by the rename worker on success. - pub rename_state: RwLock>>, - /// Per-inode async lock that serialises a rename worker against any - /// sync/pwrite-commit on the same inode. Held by the worker for the - /// entire copy+delete; held by sync around its commit. - pub rename_lock: tokio::sync::Mutex<()>, -} - -impl Inode { - /// Construct a directory inode (not yet attached to a parent's child map). - pub fn new_dir( - id: InodeId, - name: impl Into, - parent: Option>, - explicit_marker: bool, - attrs: Attrs, - ) -> Arc { - Arc::new(Self { - id, - name: name.into(), - kind: RwLock::new(InodeKind::Directory { explicit_marker }), - attrs: RwLock::new(attrs), - state: RwLock::new(InodeState::Cached), - parent, - children: RwLock::new(Children::default()), - rename_state: RwLock::new(None), - rename_lock: tokio::sync::Mutex::new(()), - }) - } - - /// Construct a regular-file inode. - pub fn new_file( - id: InodeId, - name: impl Into, - parent: Weak, - attrs: Attrs, - ) -> Arc { - Arc::new(Self { - id, - name: name.into(), - kind: RwLock::new(InodeKind::RegularFile), - attrs: RwLock::new(attrs), - state: RwLock::new(InodeState::Cached), - parent: Some(parent), - children: RwLock::new(Children::default()), - rename_state: RwLock::new(None), - rename_lock: tokio::sync::Mutex::new(()), - }) - } - - /// Construct a symlink inode. - pub fn new_symlink( - id: InodeId, - name: impl Into, - parent: Weak, - target: impl Into, - attrs: Attrs, - ) -> Arc { - Arc::new(Self { - id, - name: name.into(), - kind: RwLock::new(InodeKind::Symlink { - target: target.into(), - }), - attrs: RwLock::new(attrs), - state: RwLock::new(InodeState::Cached), - parent: Some(parent), - children: RwLock::new(Children::default()), - rename_state: RwLock::new(None), - rename_lock: tokio::sync::Mutex::new(()), - }) - } - - /// Snapshot of the inode's current rename state (None if not in flight). - pub fn rename_state(&self) -> Option> { - self.rename_state.read().clone() - } - - pub fn state(&self) -> InodeState { - *self.state.read() - } - - pub fn set_state(&self, s: InodeState) { - *self.state.write() = s; - } - - pub fn is_deleted(&self) -> bool { - matches!(self.state(), InodeState::Deleted) - } - - pub fn is_dir(&self) -> bool { - self.kind.read().is_dir() - } - - pub fn is_symlink(&self) -> bool { - self.kind.read().is_symlink() - } - - pub fn is_regular_file(&self) -> bool { - self.kind.read().is_regular_file() - } - - /// `Some(explicit_marker)` if directory, `None` otherwise. Cheap; releases - /// the lock before returning. - pub fn dir_explicit_marker(&self) -> Option { - match &*self.kind.read() { - InodeKind::Directory { explicit_marker } => Some(*explicit_marker), - _ => None, - } - } - - /// `Some(target_clone)` if symlink, `None` otherwise. - pub fn symlink_target(&self) -> Option { - match &*self.kind.read() { - InodeKind::Symlink { target } => Some(target.clone()), - _ => None, - } - } -} diff --git a/crates/s3fs-core/src/inode/listing.rs b/crates/s3fs-core/src/inode/listing.rs deleted file mode 100644 index 7bdba60..0000000 --- a/crates/s3fs-core/src/inode/listing.rs +++ /dev/null @@ -1,350 +0,0 @@ -//! Directory listing snapshot for `read-directory`. -//! -//! Per WASI Preview 2 semantics this is a *snapshot* taken at iterator-open -//! time: entries added or removed mid-iteration may be missed (this is -//! POSIX-allowed). A single call produces the full snapshot by paginating -//! `ListObjectsV2` until the cursor exhausts. - -use std::collections::HashMap; -use std::sync::Arc; -use std::time::Instant; - -use super::attrs::{Attrs, Inode}; -use super::tree::InodeTree; -use crate::backend::ListBlobsInput; -use crate::errors::{FsError, FsResult}; - -/// One entry in a directory snapshot. -#[derive(Debug, Clone)] -pub struct DirEntry { - pub name: String, - pub inode: Arc, -} - -/// Internal classification while paginating a listing. Used to decide which -/// inode kind to construct, and to give files priority over dirs when both -/// representations of the same basename appear (the geesefs convention). -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -enum EntrySrc { - File, - ExplicitDir, - ImplicitDir, -} - -impl InodeTree { - /// Snapshot the contents of `dir` for iteration. - /// - /// On the way through, opportunistically install any newly-discovered - /// children into `dir`'s child cache, and mark the listing complete so - /// later `lookup` misses can short-circuit to `NotFound`. - pub async fn snapshot_directory(&self, dir: &Arc) -> FsResult> { - if !dir.is_dir() { - return Err(FsError::NotDirectory); - } - if dir.is_deleted() { - return Err(FsError::NotFound); - } - - // The S3 prefix to list under. For the root that's just the bucket - // prefix (possibly empty). - let prefix_for_list = if dir.id == super::attrs::InodeId::ROOT { - // s3_key(root) is just the bucket_prefix; we want it slash-terminated - // unless empty. - let p = self.s3_key(dir); - if p.is_empty() { - String::new() - } else { - format!("{p}/") - } - } else { - self.s3_dir_key(dir) - }; - - let mut continuation: Option = None; - // Accumulator: dedup by basename, preferring File > ExplicitDir > ImplicitDir. - let mut acc: HashMap = HashMap::new(); - - loop { - let out = self - .backend - .list_blobs(ListBlobsInput { - prefix: &prefix_for_list, - delimiter: Some("/"), - continuation_token: continuation.as_deref(), - max_keys: None, - ..Default::default() - }) - .await?; - - for item in out.items { - let basename_part = item.key.strip_prefix(&prefix_for_list).unwrap_or(&item.key); - if basename_part.is_empty() { - // The directory's own self-marker (`dir/`); skip. - continue; - } - if let Some(name) = basename_part.strip_suffix('/') { - if name.is_empty() { - continue; - } - let attrs = Attrs { - size: 0, - etag: item.e_tag.clone(), - last_modified: item.last_modified, - content_type: None, - metadata: HashMap::new(), - fetched_at: Instant::now(), - }; - acc.entry(name.to_string()) - .or_insert((attrs, EntrySrc::ExplicitDir)); - } else { - let attrs = Attrs { - size: item.size, - etag: item.e_tag.clone(), - last_modified: item.last_modified, - content_type: None, - metadata: HashMap::new(), - fetched_at: Instant::now(), - }; - // Files take precedence over any prior dir entry. - acc.insert(basename_part.to_string(), (attrs, EntrySrc::File)); - } - } - - for cp in out.prefixes { - let basename = cp.strip_prefix(&prefix_for_list).unwrap_or(&cp); - let name = basename.strip_suffix('/').unwrap_or(basename); - if name.is_empty() { - continue; - } - acc.entry(name.to_string()) - .or_insert((Attrs::synthetic_implicit_dir(), EntrySrc::ImplicitDir)); - } - - if !out.is_truncated { - break; - } - match out.next_continuation_token { - Some(tok) => continuation = Some(tok), - None => break, // defensive - } - } - - // Pass 2: materialize inodes (cache-first). - let mut entries: Vec = Vec::with_capacity(acc.len()); - for (name, (attrs, src)) in acc { - let existing = dir.children.read().by_name.get(&name).cloned(); - let inode = match existing { - Some(i) if !i.is_deleted() => i, - _ => { - let id = self.alloc_id(); - let new_inode = match src { - EntrySrc::File => Inode::new_file(id, &name, Arc::downgrade(dir), attrs), - EntrySrc::ExplicitDir => { - Inode::new_dir(id, &name, Some(Arc::downgrade(dir)), true, attrs) - } - EntrySrc::ImplicitDir => { - Inode::new_dir(id, &name, Some(Arc::downgrade(dir)), false, attrs) - } - }; - self.attach(dir, new_inode) - } - }; - entries.push(DirEntry { name, inode }); - } - - // Mark the listing complete so future `lookup` misses can short- - // circuit to NotFound without hitting S3. - { - let mut children = dir.children.write(); - children.listing_complete = true; - children.listing_fetched_at = Some(Instant::now()); - } - - entries.sort_by(|a, b| a.name.cmp(&b.name)); - Ok(entries) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::backend::memory::MemoryBackend; - use crate::backend::{Backend, PutBlobInput}; - use crate::config::Config; - use crate::errors::FsError; - use bytes::Bytes; - use std::collections::HashMap; - - fn fresh() -> (Arc, Arc) { - let backend = Arc::new(MemoryBackend::new()); - let tree = InodeTree::new( - backend.clone() as Arc, - Arc::new(Config::default()), - ); - (backend, tree) - } - - fn fresh_with_prefix(p: &str) -> (Arc, Arc) { - let backend = Arc::new(MemoryBackend::new()); - let cfg = Config::builder().bucket_prefix(p).build(); - let tree = InodeTree::new(backend.clone() as Arc, Arc::new(cfg)); - (backend, tree) - } - - async fn put(backend: &MemoryBackend, key: &str, body: &'static [u8]) { - backend - .put_blob(PutBlobInput { - key: key.into(), - body: Bytes::from_static(body), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - } - - #[tokio::test] - async fn snapshot_root_with_files_only() { - let (backend, tree) = fresh(); - put(&backend, "a.txt", b"a").await; - put(&backend, "b.txt", b"b").await; - put(&backend, "c.txt", b"c").await; - - let snap = tree.snapshot_directory(&tree.root()).await.unwrap(); - let names: Vec<_> = snap.iter().map(|e| e.name.as_str()).collect(); - assert_eq!(names, vec!["a.txt", "b.txt", "c.txt"]); - for e in &snap { - assert!(e.inode.is_regular_file()); - } - } - - #[tokio::test] - async fn snapshot_mixes_files_and_dirs() { - let (backend, tree) = fresh(); - put(&backend, "file1.txt", b"x").await; - put(&backend, "subdir/inner.txt", b"y").await; // implicit dir - put(&backend, "explicit/", b"").await; // explicit dir marker - put(&backend, "explicit/inner.txt", b"z").await; - - let snap = tree.snapshot_directory(&tree.root()).await.unwrap(); - let names: Vec<_> = snap.iter().map(|e| e.name.as_str()).collect(); - assert_eq!(names, vec!["explicit", "file1.txt", "subdir"]); - - let by_name: HashMap<_, _> = snap - .iter() - .map(|e| (e.name.clone(), e.inode.clone())) - .collect(); - assert!(by_name["file1.txt"].is_regular_file()); - assert!(by_name["explicit"].is_dir()); - assert!(by_name["subdir"].is_dir()); - - // Note: with `delim="/"` S3 collapses both the `explicit/` marker - // object and the `explicit/inner.txt` content keys into a single - // CommonPrefix entry. From a listing alone we cannot distinguish - // explicit-via-marker from implicit-via-prefix-only — both surface - // as `explicit_marker == false`. The marker bit gets refreshed by - // a subsequent `lookup()` (which does its own HEAD on `explicit/`). - assert_eq!(by_name["explicit"].dir_explicit_marker(), Some(false)); - assert_eq!(by_name["subdir"].dir_explicit_marker(), Some(false)); - - // Prove the upgrade: looking up "explicit" individually issues a HEAD - // on the marker and (if found) yields explicit_marker=true. We need - // to bypass the cache to force the lookup to hit the backend, so use - // a fresh tree wired to the same backend. - let tree2 = InodeTree::new( - backend.clone() as Arc, - Arc::new(Config::default()), - ); - let upgraded = tree2.lookup(&tree2.root(), "explicit").await.unwrap(); - assert_eq!(upgraded.dir_explicit_marker(), Some(true)); - } - - #[tokio::test] - async fn snapshot_deduplicates_explicit_marker_and_implicit_prefix() { - let (backend, tree) = fresh(); - // Both an explicit `dir/` marker AND child objects under `dir/`. - put(&backend, "dir/", b"").await; - put(&backend, "dir/a", b"").await; - put(&backend, "dir/b", b"").await; - - let snap = tree.snapshot_directory(&tree.root()).await.unwrap(); - // Should appear exactly once, classified as explicit (file path takes - // precedence in the dedup rule, but here both candidates are dir-like - // so explicit wins via insertion order). - assert_eq!(snap.len(), 1); - assert_eq!(snap[0].name, "dir"); - assert!(snap[0].inode.is_dir()); - } - - #[tokio::test] - async fn snapshot_subdir_only_lists_immediate_children() { - let (backend, tree) = fresh(); - put(&backend, "dir/a", b"x").await; - put(&backend, "dir/b", b"y").await; - put(&backend, "dir/sub/c", b"z").await; - put(&backend, "outside.txt", b"").await; - - let dir = tree.lookup(&tree.root(), "dir").await.unwrap(); - let snap = tree.snapshot_directory(&dir).await.unwrap(); - let names: Vec<_> = snap.iter().map(|e| e.name.as_str()).collect(); - assert_eq!(names, vec!["a", "b", "sub"]); - } - - #[tokio::test] - async fn snapshot_marks_listing_complete_so_lookup_short_circuits() { - let (backend, tree) = fresh(); - put(&backend, "only.txt", b"x").await; - - // First, snapshot. - let _ = tree.snapshot_directory(&tree.root()).await.unwrap(); - assert!(tree.root().children.read().listing_complete); - - // Then look up something we know isn't there. Even if we delete the - // backend entirely, the lookup should NOT hit it because the parent's - // listing is marked complete. - // (Easiest way to verify: drop the backend's state.) - // We can't actually drop the backend here since it's shared; instead, - // rely on the assertion that the cache is consulted first. We can - // still verify the short-circuit by observing that no extra ops are - // needed for an obviously-absent name. - let res = tree.lookup(&tree.root(), "does-not-exist").await; - assert!(matches!(res, Err(FsError::NotFound))); - - // And the put-known child must still be findable through the cache. - let _ = backend; // keep alive - let known = tree.lookup(&tree.root(), "only.txt").await.unwrap(); - assert!(known.is_regular_file()); - } - - #[tokio::test] - async fn snapshot_with_bucket_prefix() { - let (backend, tree) = fresh_with_prefix("data"); - put(&backend, "data/a.txt", b"x").await; - put(&backend, "data/sub/b.txt", b"y").await; - put(&backend, "outside/x", b"z").await; - - let snap = tree.snapshot_directory(&tree.root()).await.unwrap(); - let names: Vec<_> = snap.iter().map(|e| e.name.as_str()).collect(); - // Only entries under the prefix should appear; "outside/x" must not. - assert_eq!(names, vec!["a.txt", "sub"]); - } - - #[tokio::test] - async fn snapshot_empty_directory() { - let (_backend, tree) = fresh(); - let snap = tree.snapshot_directory(&tree.root()).await.unwrap(); - assert!(snap.is_empty()); - assert!(tree.root().children.read().listing_complete); - } - - #[tokio::test] - async fn snapshot_rejects_non_directory() { - let (backend, tree) = fresh(); - put(&backend, "file.txt", b"x").await; - let f = tree.lookup(&tree.root(), "file.txt").await.unwrap(); - assert!(matches!( - tree.snapshot_directory(&f).await, - Err(FsError::NotDirectory) - )); - } -} diff --git a/crates/s3fs-core/src/inode/lookup.rs b/crates/s3fs-core/src/inode/lookup.rs deleted file mode 100644 index f58a294..0000000 --- a/crates/s3fs-core/src/inode/lookup.rs +++ /dev/null @@ -1,599 +0,0 @@ -//! Race-three lookup and openat-style path traversal. -//! -//! `lookup` resolves a single child of a directory using the GeeseFS -//! `LookUpInodeMaybeDir` strategy: race a `HeadObject(name)`, a -//! `HeadObject(name/)`, and a `ListObjectsV2(prefix=name/, max_keys=1)` in -//! parallel. Priority: regular file > explicit directory > implicit directory. -//! -//! `lookup_at` walks a relative POSIX path one component at a time, using -//! `lookup` per step, and enforces preopen containment: `..` cannot escape -//! the [`InodeTree`]'s root (which corresponds to the preopen). - -use std::collections::HashMap; -use std::sync::{Arc, Weak}; -use std::time::Instant; - -use super::attrs::{Attrs, Inode, InodeKind}; -use super::tree::InodeTree; -use crate::backend::{BlobMeta, ListBlobsInput}; -use crate::errors::{FsError, FsResult}; -use crate::path; - -/// Metadata key (after S3 strips the `x-amz-meta-` prefix) marking a symlink -/// object. Body of such an object is the literal target path. -pub const SYMLINK_METADATA_KEY: &str = "s3wasifs-type"; -pub const SYMLINK_METADATA_VALUE: &str = "symlink"; - -fn meta_indicates_symlink(meta: &HashMap) -> bool { - meta.get(SYMLINK_METADATA_KEY) - .map(|s| s == SYMLINK_METADATA_VALUE) - .unwrap_or(false) -} - -fn attrs_from_head(head: &BlobMeta) -> Attrs { - // If we previously persisted an mtime via `set_times`, prefer that over - // S3's `LastModified` so guests see the time they explicitly set. - let last_modified = head - .metadata - .get(crate::fs::METADATA_MTIME_KEY) - .and_then(|s| crate::fs::meta_to_systemtime(s)) - .unwrap_or(head.last_modified); - Attrs { - size: head.size, - etag: head.e_tag.clone(), - last_modified, - content_type: head.content_type.clone(), - metadata: head.metadata.clone(), - fetched_at: Instant::now(), - } -} - -impl InodeTree { - /// Look up `name` as a child of `parent`. Cache-first; on miss, race three - /// backend ops in parallel. - /// - /// Returns the resolved child inode (file, dir, or symlink). Stale - /// (`Deleted`) cache entries are treated as misses. - pub async fn lookup(&self, parent: &Arc, name: &str) -> FsResult> { - if !parent.is_dir() { - return Err(FsError::NotDirectory); - } - if parent.is_deleted() { - return Err(FsError::NotFound); - } - path::validate_segment(name)?; - - // 1) Cache lookup. Hold the read lock only for the duration of the - // probe; release before any await. - { - let children = parent.children.read(); - if let Some(child) = children.by_name.get(name) { - if !child.is_deleted() { - return Ok(child.clone()); - } - } else if children.listing_complete { - // Directory has been fully enumerated and the child is absent. - return Err(FsError::NotFound); - } - } - - // 2) Race three. - let parent_key = self.s3_key(parent); - let file_key = if parent_key.is_empty() { - name.to_string() - } else { - format!("{parent_key}/{name}") - }; - let dir_key = path::dir_marker_key(&file_key); - - let (file_res, dir_res, list_res) = tokio::join!( - self.backend.head_blob(&file_key), - self.backend.head_blob(&dir_key), - self.backend.list_blobs(ListBlobsInput { - prefix: &dir_key, - delimiter: None, - max_keys: Some(1), - ..Default::default() - }), - ); - - // Re-check cache before insert: another concurrent lookup may have - // installed the same child while we were awaiting. - { - let children = parent.children.read(); - if let Some(child) = children.by_name.get(name) { - if !child.is_deleted() { - return Ok(child.clone()); - } - } - } - - // Priority order: file > explicit dir > implicit dir. - if let Ok(meta) = file_res { - let attrs = attrs_from_head(&meta); - let kind = if meta_indicates_symlink(&meta.metadata) { - // For a symlink, the target body lives in the object payload. - // Issue a follow-up GET so the inode is fully populated. - let body = self.backend.get_blob(&file_key, None).await?; - let target = String::from_utf8(body.body.to_vec()) - .map_err(|_| FsError::IllegalByteSequence)?; - path::validate_symlink_target(&target)?; - InodeKind::Symlink { target } - } else { - InodeKind::RegularFile - }; - let id = self.alloc_id(); - let child = match kind { - InodeKind::Symlink { target } => { - Inode::new_symlink(id, name, Arc::downgrade(parent), target, attrs) - } - _ => Inode::new_file(id, name, Arc::downgrade(parent), attrs), - }; - return Ok(self.attach(parent, child)); - } - - if let Ok(meta) = dir_res { - let attrs = attrs_from_head(&meta); - let id = self.alloc_id(); - let child = Inode::new_dir( - id, - name, - Some(Arc::downgrade(parent)), - /* explicit_marker */ true, - attrs, - ); - return Ok(self.attach(parent, child)); - } - - if let Ok(out) = list_res { - if !out.items.is_empty() || !out.prefixes.is_empty() { - let id = self.alloc_id(); - let child = Inode::new_dir( - id, - name, - Some(Arc::downgrade(parent)), - /* explicit_marker */ false, - Attrs::synthetic_implicit_dir(), - ); - return Ok(self.attach(parent, child)); - } - // Empty listing AND both heads NotFound → genuinely absent. - } - - Err(FsError::NotFound) - } - - /// Walk `rel_path` from `start`, calling `lookup` for each non-special - /// component. `.` is skipped; `..` pops the current node, but cannot - /// escape the tree root (which is the preopen root). - /// - /// Symlinks are followed transparently for **intermediate** components. - /// The final named component is also followed (POSIX default). To - /// suppress follow on the final component (POSIX `O_NOFOLLOW` / - /// WASI `path-flags::symlink-follow` cleared), use [`Self::lookup_at_no_follow`]. - pub async fn lookup_at(&self, start: &Arc, rel_path: &str) -> FsResult> { - self.lookup_at_inner(start, rel_path, /* follow_last */ true, 0) - .await - } - - /// Like `lookup_at` but does NOT follow a symlink at the final named - /// component. Returns the symlink inode itself if the leaf is a symlink. - /// Used by `readlink_at` and by `open_at` when `path-flags::symlink-follow` - /// is cleared (the latter then maps to `error-code::loop`). - pub async fn lookup_at_no_follow( - &self, - start: &Arc, - rel_path: &str, - ) -> FsResult> { - self.lookup_at_inner(start, rel_path, /* follow_last */ false, 0) - .await - } - - /// Internal recursive implementation. Tracks `depth` to enforce - /// `Config::max_symlink_depth` (default 40, matching Linux's `MAXSYMLINKS`). - async fn lookup_at_inner( - &self, - start: &Arc, - rel_path: &str, - follow_last: bool, - depth: u32, - ) -> FsResult> { - if rel_path.starts_with('/') { - return Err(FsError::NotPermitted); - } - if depth > self.config.max_symlink_depth { - return Err(FsError::Loop); - } - let mut current = start.clone(); - let root = self.root(); - - let segments: Vec<&str> = rel_path.split('/').collect(); - // Index of the rightmost "named" segment (not "", ".", or ".."). The - // `follow_last` flag only governs that one — all earlier symlinks are - // always followed. - let last_named = segments - .iter() - .rposition(|s| !s.is_empty() && *s != "." && *s != ".."); - - for (i, seg) in segments.iter().enumerate() { - if seg.is_empty() || *seg == "." { - continue; - } - if *seg == ".." { - if Arc::ptr_eq(¤t, &root) { - return Err(FsError::NotPermitted); - } - let parent = current - .parent - .as_ref() - .and_then(Weak::upgrade) - .ok_or(FsError::NotPermitted)?; - current = parent; - continue; - } - if !current.is_dir() { - return Err(FsError::NotDirectory); - } - current = self.lookup(¤t, seg).await?; - - // Symlink follow: always for intermediate components, and for - // the final named component iff `follow_last`. - let is_last_named = Some(i) == last_named; - let should_follow = !is_last_named || follow_last; - if should_follow { - if let Some(target) = current.symlink_target() { - // Resolve relative to the symlink's parent (or root for - // absolute targets). Recurse with depth + 1 (Box::pin so - // the future has a known size). - let (resolution_base, target_relative) = - if let Some(stripped) = target.strip_prefix('/') { - (root.clone(), stripped.to_string()) - } else { - let parent = current - .parent - .as_ref() - .and_then(Weak::upgrade) - .unwrap_or_else(|| root.clone()); - (parent, target) - }; - current = Box::pin(self.lookup_at_inner( - &resolution_base, - &target_relative, - /* follow_last */ true, - depth + 1, - )) - .await?; - } - } - } - Ok(current) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::backend::memory::MemoryBackend; - use crate::backend::{Backend, PutBlobInput}; - use crate::config::Config; - use bytes::Bytes; - use std::collections::HashMap; - - fn fresh() -> (Arc, Arc) { - let backend = Arc::new(MemoryBackend::new()); - let tree = InodeTree::new( - backend.clone() as Arc, - Arc::new(Config::default()), - ); - (backend, tree) - } - - fn fresh_with_prefix(prefix: &str) -> (Arc, Arc) { - let backend = Arc::new(MemoryBackend::new()); - let cfg = Config::builder().bucket_prefix(prefix).build(); - let tree = InodeTree::new(backend.clone() as Arc, Arc::new(cfg)); - (backend, tree) - } - - async fn put(backend: &MemoryBackend, key: &str, body: &'static [u8]) { - backend - .put_blob(PutBlobInput { - key: key.into(), - body: Bytes::from_static(body), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - } - - async fn put_with_meta( - backend: &MemoryBackend, - key: &str, - body: &'static [u8], - metadata: HashMap, - ) { - backend - .put_blob(PutBlobInput { - key: key.into(), - body: Bytes::from_static(body), - metadata, - content_type: None, - }) - .await - .unwrap(); - } - - #[tokio::test] - async fn lookup_finds_regular_file() { - let (backend, tree) = fresh(); - put(&backend, "hello.txt", b"hi").await; - - let root = tree.root(); - let child = tree.lookup(&root, "hello.txt").await.unwrap(); - assert!(child.is_regular_file()); - assert_eq!(tree.full_key(&child), "hello.txt"); - assert_eq!(child.attrs.read().size, 2); - } - - #[tokio::test] - async fn lookup_finds_explicit_directory_marker() { - let (backend, tree) = fresh(); - // A geesefs-style mkdir: zero-byte object whose key ends in '/'. - put(&backend, "subdir/", b"").await; - - let root = tree.root(); - let child = tree.lookup(&root, "subdir").await.unwrap(); - assert!(child.is_dir()); - assert_eq!(child.dir_explicit_marker(), Some(true)); - } - - #[tokio::test] - async fn lookup_finds_implicit_directory_via_prefix() { - let (backend, tree) = fresh(); - // No marker; only a child key exists. Directory is implicit. - put(&backend, "implicit/file.txt", b"x").await; - - let root = tree.root(); - let child = tree.lookup(&root, "implicit").await.unwrap(); - assert!(child.is_dir()); - assert_eq!(child.dir_explicit_marker(), Some(false)); - } - - #[tokio::test] - async fn lookup_returns_not_found_when_absent() { - let (_backend, tree) = fresh(); - let root = tree.root(); - assert!(matches!( - tree.lookup(&root, "missing").await, - Err(FsError::NotFound) - )); - } - - #[tokio::test] - async fn file_wins_over_directory_when_both_present() { - let (backend, tree) = fresh(); - // Pathological: both `foo` (file) and `foo/bar` (dir contents) exist. - put(&backend, "foo", b"file body").await; - put(&backend, "foo/bar", b"x").await; - - let root = tree.root(); - let child = tree.lookup(&root, "foo").await.unwrap(); - assert!(child.is_regular_file()); - assert_eq!(child.attrs.read().size, 9); - } - - #[tokio::test] - async fn lookup_caches_then_skips_backend_on_repeat() { - let (backend, tree) = fresh(); - put(&backend, "hello.txt", b"hi").await; - let root = tree.root(); - - let a = tree.lookup(&root, "hello.txt").await.unwrap(); - let b = tree.lookup(&root, "hello.txt").await.unwrap(); - // Same Arc → cached - assert!(Arc::ptr_eq(&a, &b)); - } - - #[tokio::test] - async fn lookup_rejects_not_a_directory_parent() { - let (backend, tree) = fresh(); - put(&backend, "file.txt", b"x").await; - let root = tree.root(); - let f = tree.lookup(&root, "file.txt").await.unwrap(); - assert!(matches!( - tree.lookup(&f, "child").await, - Err(FsError::NotDirectory) - )); - } - - #[tokio::test] - async fn lookup_validates_segment() { - let (_backend, tree) = fresh(); - let root = tree.root(); - assert!(matches!( - tree.lookup(&root, "..").await, - Err(FsError::Invalid(_)) - )); - assert!(matches!( - tree.lookup(&root, "with\0nul").await, - Err(FsError::IllegalByteSequence) - )); - } - - #[tokio::test] - async fn lookup_with_bucket_prefix_uses_correct_keys() { - let (backend, tree) = fresh_with_prefix("data"); - put(&backend, "data/file.txt", b"hi").await; - - let root = tree.root(); - let f = tree.lookup(&root, "file.txt").await.unwrap(); - assert!(f.is_regular_file()); - assert_eq!(tree.s3_key(&f), "data/file.txt"); - } - - #[tokio::test] - async fn lookup_detects_symlink_via_metadata_and_fetches_target() { - let (backend, tree) = fresh(); - let mut meta = HashMap::new(); - meta.insert(SYMLINK_METADATA_KEY.into(), SYMLINK_METADATA_VALUE.into()); - put_with_meta(&backend, "link", b"target.txt", meta).await; - - let root = tree.root(); - let l = tree.lookup(&root, "link").await.unwrap(); - assert!(l.is_symlink()); - assert_eq!(l.symlink_target().as_deref(), Some("target.txt")); - } - - #[tokio::test] - async fn lookup_at_walks_components() { - let (backend, tree) = fresh(); - put(&backend, "a/b/c.txt", b"deep").await; - - let root = tree.root(); - let c = tree.lookup_at(&root, "a/b/c.txt").await.unwrap(); - assert!(c.is_regular_file()); - assert_eq!(tree.full_key(&c), "a/b/c.txt"); - } - - #[tokio::test] - async fn lookup_at_skips_dot_and_pops_dotdot() { - let (backend, tree) = fresh(); - put(&backend, "a/b/c.txt", b"x").await; - put(&backend, "a/sibling", b"y").await; - - let root = tree.root(); - let r = tree.lookup_at(&root, "a/./b/../sibling").await.unwrap(); - assert!(r.is_regular_file()); - assert_eq!(tree.full_key(&r), "a/sibling"); - } - - #[tokio::test] - async fn lookup_at_rejects_dotdot_escape_above_root() { - let (_backend, tree) = fresh(); - let root = tree.root(); - assert!(matches!( - tree.lookup_at(&root, "..").await, - Err(FsError::NotPermitted) - )); - } - - #[tokio::test] - async fn lookup_at_rejects_absolute_path() { - let (_backend, tree) = fresh(); - let root = tree.root(); - assert!(matches!( - tree.lookup_at(&root, "/etc/passwd").await, - Err(FsError::NotPermitted) - )); - } - - #[tokio::test] - async fn lookup_at_empty_returns_start() { - let (_backend, tree) = fresh(); - let root = tree.root(); - let r = tree.lookup_at(&root, "").await.unwrap(); - assert!(Arc::ptr_eq(&r, &root)); - } - - #[tokio::test] - async fn lookup_at_follows_symlink_intermediate_component() { - let (backend, tree) = fresh(); - // /target/file.txt - put(&backend, "target/file.txt", b"hello").await; - // /link → "target" (a symlink to a dir) - let mut meta = HashMap::new(); - meta.insert(SYMLINK_METADATA_KEY.into(), SYMLINK_METADATA_VALUE.into()); - put_with_meta(&backend, "link", b"target", meta).await; - - // lookup "link/file.txt" — intermediate "link" is followed. - let resolved = tree.lookup_at(&tree.root(), "link/file.txt").await.unwrap(); - assert!(resolved.is_regular_file()); - assert_eq!(tree.full_key(&resolved), "target/file.txt"); - } - - #[tokio::test] - async fn lookup_at_follows_symlink_at_leaf_by_default() { - let (backend, tree) = fresh(); - put(&backend, "target.txt", b"yo").await; - let mut meta = HashMap::new(); - meta.insert(SYMLINK_METADATA_KEY.into(), SYMLINK_METADATA_VALUE.into()); - put_with_meta(&backend, "link.txt", b"target.txt", meta).await; - - let resolved = tree.lookup_at(&tree.root(), "link.txt").await.unwrap(); - assert!(resolved.is_regular_file()); - assert_eq!(tree.full_key(&resolved), "target.txt"); - } - - #[tokio::test] - async fn lookup_at_no_follow_returns_symlink_at_leaf() { - let (backend, tree) = fresh(); - put(&backend, "target.txt", b"yo").await; - let mut meta = HashMap::new(); - meta.insert(SYMLINK_METADATA_KEY.into(), SYMLINK_METADATA_VALUE.into()); - put_with_meta(&backend, "link.txt", b"target.txt", meta).await; - - let resolved = tree - .lookup_at_no_follow(&tree.root(), "link.txt") - .await - .unwrap(); - assert!(resolved.is_symlink()); - assert_eq!(resolved.symlink_target().as_deref(), Some("target.txt")); - } - - #[tokio::test] - async fn lookup_at_symlink_cycle_returns_loop() { - let (backend, tree) = fresh(); - let meta = || { - let mut m = HashMap::new(); - m.insert(SYMLINK_METADATA_KEY.into(), SYMLINK_METADATA_VALUE.into()); - m - }; - // a → b, b → a (mutual cycle) - put_with_meta(&backend, "a", b"b", meta()).await; - put_with_meta(&backend, "b", b"a", meta()).await; - - let r = tree.lookup_at(&tree.root(), "a").await; - assert!(matches!(r, Err(FsError::Loop)), "expected Loop, got {r:?}"); - } - - #[tokio::test] - async fn lookup_at_symlink_to_absolute_path_resolves_from_root() { - let (backend, tree) = fresh(); - put(&backend, "deep/target.txt", b"x").await; - let mut meta = HashMap::new(); - meta.insert(SYMLINK_METADATA_KEY.into(), SYMLINK_METADATA_VALUE.into()); - put_with_meta(&backend, "shortcut", b"/deep/target.txt", meta).await; - - let r = tree.lookup_at(&tree.root(), "shortcut").await.unwrap(); - assert!(r.is_regular_file()); - assert_eq!(tree.full_key(&r), "deep/target.txt"); - } - - #[tokio::test] - async fn lookup_at_symlink_target_escape_rejected() { - let (backend, tree) = fresh(); - let mut meta = HashMap::new(); - meta.insert(SYMLINK_METADATA_KEY.into(), SYMLINK_METADATA_VALUE.into()); - put_with_meta(&backend, "evil", b"../../etc/passwd", meta).await; - - let r = tree.lookup_at(&tree.root(), "evil").await; - assert!( - matches!(r, Err(FsError::NotPermitted)), - "expected NotPermitted, got {r:?}" - ); - } - - #[tokio::test] - async fn lookup_at_dotdot_within_subtree_pops_to_parent() { - let (backend, tree) = fresh(); - put(&backend, "a/b/c.txt", b"x").await; - let root = tree.root(); - // open a/b first, then via that descriptor go .. - let a = tree.lookup(&root, "a").await.unwrap(); - let b = tree.lookup(&a, "b").await.unwrap(); - // From b, "../" should land back at a. - let popped = tree.lookup_at(&b, "..").await.unwrap(); - assert!(Arc::ptr_eq(&popped, &a)); - } -} diff --git a/crates/s3fs-core/src/inode/mod.rs b/crates/s3fs-core/src/inode/mod.rs deleted file mode 100644 index 3490c15..0000000 --- a/crates/s3fs-core/src/inode/mod.rs +++ /dev/null @@ -1,12 +0,0 @@ -//! Inode layer: in-memory tree, attribute cache, race-three lookup, and -//! directory listing snapshots. - -pub mod attrs; -pub mod listing; -pub mod lookup; -pub mod tree; - -pub use attrs::{Attrs, Children, Inode, InodeId, InodeKind, InodeState}; -pub use listing::DirEntry; -pub use lookup::{SYMLINK_METADATA_KEY, SYMLINK_METADATA_VALUE}; -pub use tree::InodeTree; diff --git a/crates/s3fs-core/src/inode/tree.rs b/crates/s3fs-core/src/inode/tree.rs deleted file mode 100644 index 85717dc..0000000 --- a/crates/s3fs-core/src/inode/tree.rs +++ /dev/null @@ -1,342 +0,0 @@ -//! `InodeTree` — the root container that owns every live `Inode` and exposes -//! the navigation primitives (`root`, `get`, `full_key`, `s3_key`). -//! -//! Lookup, openat traversal, and directory snapshots live in sibling modules -//! (`lookup.rs`, `listing.rs`) as additional `impl InodeTree` blocks. - -use std::collections::HashMap; -use std::sync::atomic::{AtomicU64, Ordering}; -use std::sync::{Arc, Weak}; -use std::time::SystemTime; - -use parking_lot::RwLock; - -use super::attrs::{Attrs, Children, Inode, InodeId, InodeKind, InodeState}; -use crate::backend::Backend; -use crate::config::Config; -use crate::path; - -/// In-memory inode cache, parameterised over a `Backend`. -#[derive(Debug)] -pub struct InodeTree { - by_id: RwLock>>, - next_id: AtomicU64, - pub(crate) backend: Arc, - pub(crate) config: Arc, -} - -impl InodeTree { - /// Build a new tree rooted at the bucket prefix specified in `config`. - pub fn new(backend: Arc, config: Arc) -> Arc { - let tree = Arc::new(Self { - by_id: RwLock::new(HashMap::new()), - // start counter at 2 — id 1 is reserved for root - next_id: AtomicU64::new(2), - backend, - config, - }); - let root = Arc::new(Inode { - id: InodeId::ROOT, - name: String::new(), - kind: RwLock::new(InodeKind::Directory { - explicit_marker: false, - }), - attrs: RwLock::new(Attrs { - size: 0, - etag: String::new(), - last_modified: SystemTime::UNIX_EPOCH, - content_type: None, - metadata: HashMap::new(), - fetched_at: std::time::Instant::now(), - }), - state: RwLock::new(InodeState::Cached), - parent: None, - children: RwLock::new(Children::default()), - rename_state: RwLock::new(None), - rename_lock: tokio::sync::Mutex::new(()), - }); - tree.by_id.write().insert(InodeId::ROOT, root); - tree - } - - pub fn root(&self) -> Arc { - self.by_id - .read() - .get(&InodeId::ROOT) - .expect("root inode always present") - .clone() - } - - pub fn get(&self, id: InodeId) -> Option> { - self.by_id.read().get(&id).cloned() - } - - pub(crate) fn alloc_id(&self) -> InodeId { - let n = self.next_id.fetch_add(1, Ordering::Relaxed); - InodeId(std::num::NonZeroU64::new(n).expect("inode counter overflow")) - } - - /// Build the bucket-relative (i.e. mount-relative) key for an inode by - /// walking parent pointers and joining basenames with `/`. - /// - /// Returns the empty string for the root inode. - pub fn full_key(&self, inode: &Inode) -> String { - // Walk parents, accumulating basenames in reverse. - let mut segments: Vec = Vec::new(); - if !inode.name.is_empty() { - segments.push(inode.name.clone()); - } - let mut cur = inode.parent.as_ref().and_then(Weak::upgrade); - while let Some(p) = cur { - if !p.name.is_empty() { - segments.push(p.name.clone()); - } - cur = p.parent.as_ref().and_then(Weak::upgrade); - } - segments.reverse(); - segments.join("/") - } - - /// The wire key for this inode's object — `full_key` prepended with the - /// configured `bucket_prefix`. - pub fn s3_key(&self, inode: &Inode) -> String { - path::join_prefix(&self.config.bucket_prefix, &self.full_key(inode)) - } - - /// Wire key honoring an in-flight async rename: returns the OLD key - /// while the rename worker is still propagating bytes from old → new in - /// S3, so reads/writes resolve to the object that actually exists. - /// Once the worker clears `rename_state`, falls back to `s3_key`. - pub fn current_s3_key(&self, inode: &Inode) -> String { - if let Some(s) = inode.rename_state() { - return s.old_key.clone(); - } - self.s3_key(inode) - } - - /// The directory-marker form (`s3_key(inode) + "/"`) for use with - /// `mkdir`-style explicit directory objects. - pub fn s3_dir_key(&self, inode: &Inode) -> String { - path::dir_marker_key(&self.s3_key(inode)) - } - - /// Atomically check-or-insert. If the parent already holds a non-deleted - /// child with the same name, return the existing one (the freshly-built - /// candidate is dropped). Otherwise install the candidate and return it. - /// - /// This is what makes concurrent `lookup`/`snapshot_directory` calls for - /// the same name safe — the loser's allocated `InodeId` is wasted but the - /// tree stays single-rooted per name. - pub(crate) fn attach(&self, parent: &Arc, child: Arc) -> Arc { - debug_assert!(parent.is_dir(), "parent must be a directory"); - debug_assert!(!child.name.is_empty(), "child must have a name"); - let mut children = parent.children.write(); - if let Some(existing) = children.by_name.get(&child.name) { - if !existing.is_deleted() { - return existing.clone(); - } - } - children.by_name.insert(child.name.clone(), child.clone()); - drop(children); - self.by_id.write().insert(child.id, child.clone()); - child - } - - /// Remove an inode from the by-id table and from its parent's children - /// map. Marks state `Deleted`. Does NOT recurse — callers handle - /// directory tree teardown themselves. - /// - /// Reserved for the upcoming unlink/rmdir paths; currently exercised by - /// tests only. - #[allow(dead_code)] - pub(crate) fn detach(&self, inode: &Arc) { - inode.set_state(InodeState::Deleted); - if let Some(parent) = inode.parent.as_ref().and_then(Weak::upgrade) { - parent.children.write().by_name.remove(&inode.name); - } - self.by_id.write().remove(&inode.id); - } - - /// Total number of live inodes (root + everything below). - pub fn len(&self) -> usize { - self.by_id.read().len() - } - - pub fn is_empty(&self) -> bool { - self.len() == 0 - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::backend::memory::MemoryBackend; - use crate::config::Config; - - fn fresh_tree() -> Arc { - let backend: Arc = Arc::new(MemoryBackend::new()); - let cfg = Arc::new(Config::default()); - InodeTree::new(backend, cfg) - } - - fn fresh_tree_with_prefix(prefix: &str) -> Arc { - let backend: Arc = Arc::new(MemoryBackend::new()); - let cfg = Arc::new(Config::builder().bucket_prefix(prefix).build()); - InodeTree::new(backend, cfg) - } - - #[test] - fn root_id_is_one_and_stored() { - let t = fresh_tree(); - let r = t.root(); - assert_eq!(r.id, InodeId::ROOT); - assert_eq!(r.id.get(), 1); - assert_eq!(t.len(), 1); - assert!(r.is_dir()); - assert_eq!(t.full_key(&r), ""); - } - - #[test] - fn alloc_id_is_monotonic_and_skips_root() { - let t = fresh_tree(); - let a = t.alloc_id(); - let b = t.alloc_id(); - assert_eq!(a.get(), 2); - assert_eq!(b.get(), 3); - assert_ne!(a, InodeId::ROOT); - } - - #[test] - fn attach_then_detach_a_file() { - let t = fresh_tree(); - let root = t.root(); - let id = t.alloc_id(); - let attrs = Attrs { - size: 5, - etag: "etag1".into(), - last_modified: SystemTime::UNIX_EPOCH, - content_type: None, - metadata: HashMap::new(), - fetched_at: std::time::Instant::now(), - }; - let child = Inode::new_file(id, "hello.txt", Arc::downgrade(&root), attrs); - t.attach(&root, child.clone()); - - assert_eq!(t.len(), 2); - assert_eq!(t.full_key(&child), "hello.txt"); - assert!(root.children.read().by_name.contains_key("hello.txt")); - - t.detach(&child); - assert_eq!(t.len(), 1); - assert!(child.is_deleted()); - assert!(!root.children.read().by_name.contains_key("hello.txt")); - } - - #[test] - fn full_key_walks_parent_chain() { - let t = fresh_tree(); - let root = t.root(); - - // root → "dir" → "sub" → "file.txt" - let attrs = || Attrs { - size: 0, - etag: String::new(), - last_modified: SystemTime::UNIX_EPOCH, - content_type: None, - metadata: HashMap::new(), - fetched_at: std::time::Instant::now(), - }; - let dir = Inode::new_dir( - t.alloc_id(), - "dir", - Some(Arc::downgrade(&root)), - false, - attrs(), - ); - t.attach(&root, dir.clone()); - let sub = Inode::new_dir( - t.alloc_id(), - "sub", - Some(Arc::downgrade(&dir)), - false, - attrs(), - ); - t.attach(&dir, sub.clone()); - let file = Inode::new_file(t.alloc_id(), "file.txt", Arc::downgrade(&sub), attrs()); - t.attach(&sub, file.clone()); - - assert_eq!(t.full_key(&file), "dir/sub/file.txt"); - assert_eq!(t.full_key(&sub), "dir/sub"); - assert_eq!(t.full_key(&dir), "dir"); - assert_eq!(t.full_key(&root), ""); - } - - #[test] - fn s3_key_applies_bucket_prefix() { - let t = fresh_tree_with_prefix("data"); - let root = t.root(); - let f = Inode::new_file( - t.alloc_id(), - "x.txt", - Arc::downgrade(&root), - Attrs { - size: 0, - etag: String::new(), - last_modified: SystemTime::UNIX_EPOCH, - content_type: None, - metadata: HashMap::new(), - fetched_at: std::time::Instant::now(), - }, - ); - t.attach(&root, f.clone()); - assert_eq!(t.s3_key(&f), "data/x.txt"); - assert_eq!(t.s3_dir_key(&f), "data/x.txt/"); - assert_eq!(t.s3_key(&root), "data"); - } - - #[test] - fn s3_key_with_empty_prefix_is_just_full_key() { - let t = fresh_tree(); - let root = t.root(); - let f = Inode::new_file( - t.alloc_id(), - "x", - Arc::downgrade(&root), - Attrs { - size: 0, - etag: String::new(), - last_modified: SystemTime::UNIX_EPOCH, - content_type: None, - metadata: HashMap::new(), - fetched_at: std::time::Instant::now(), - }, - ); - t.attach(&root, f.clone()); - assert_eq!(t.s3_key(&f), "x"); - assert_eq!(t.s3_dir_key(&f), "x/"); - assert_eq!(t.s3_key(&root), ""); - } - - #[test] - fn detached_inode_kind_unchanged_but_state_deleted() { - let t = fresh_tree(); - let root = t.root(); - let f = Inode::new_file( - t.alloc_id(), - "x", - Arc::downgrade(&root), - Attrs { - size: 0, - etag: String::new(), - last_modified: SystemTime::UNIX_EPOCH, - content_type: None, - metadata: HashMap::new(), - fetched_at: std::time::Instant::now(), - }, - ); - t.attach(&root, f.clone()); - t.detach(&f); - assert!(f.is_deleted()); - assert!(f.is_regular_file()); // kind preserved for stale-handle errors - } -} diff --git a/crates/s3fs-core/src/lib.rs b/crates/s3fs-core/src/lib.rs index afd31a2..130adff 100644 --- a/crates/s3fs-core/src/lib.rs +++ b/crates/s3fs-core/src/lib.rs @@ -1,21 +1,33 @@ -//! `s3fs-core` — async S3-backed filesystem engine. +//! `s3fs-core` — a ZFS-style, Merkle-anchored filesystem over S3. //! -//! See the workspace README and the design plan for context. The Compatibility -//! Matrix in the README is the authoritative user-facing semantic contract. +//! Data lives in immutable, encrypted blocks packed into slab objects; the +//! whole filesystem hangs off a single signed, hash-chained root record written +//! with a conditional PUT under S3 Object Lock. Verifying that record +//! transitively verifies every byte beneath it, and the chain of records is +//! what makes a rollback detectable rather than merely unlikely. +//! +//! ```text +//! Fs POSIX semantics: paths, handles, directories +//! │ +//! store::Store transaction groups, the root chain, the object set +//! │ +//! Backend S3, S3-compatible, or an in-memory fake +//! ``` +//! +//! See [`store`] for the on-disk format and [`crypto`] for the key hierarchy. pub mod backend; -pub mod buffer; pub mod config; +pub mod crypto; pub mod errors; -pub mod flusher; pub mod fs; pub mod inode; -pub mod mpu; pub mod path; -pub mod rename; +pub mod store; -pub use buffer::{BufferPool, PartBuf, PartKey, PartState, RangeSet}; pub use config::Config; -pub use errors::FsError; -pub use fs::{FileHandle, Fs, HandleId, OpenFlags}; -pub use inode::{DirEntry, Inode, InodeId, InodeKind, InodeState, InodeTree}; +pub use crypto::MasterSecret; +pub use errors::{FsError, FsResult}; +pub use fs::{FileHandle, Fs, HandleId, OpenFlags, SnapshotFs, SnapshotInfo}; +pub use inode::{Attrs, DirEntry, Inode, InodeId, InodeKind}; +pub use store::{RootRecord, Store}; diff --git a/crates/s3fs-core/src/mpu.rs b/crates/s3fs-core/src/mpu.rs deleted file mode 100644 index 03a2db4..0000000 --- a/crates/s3fs-core/src/mpu.rs +++ /dev/null @@ -1,632 +0,0 @@ -//! `MpuState` — per-file multipart-upload state machine. -//! -//! Tracks which parts have been uploaded for an in-flight MPU and which are -//! still missing (and therefore need either a fresh `UploadPart` or a -//! server-side `UploadPartCopy` from the source object during commit). -//! -//! The load-bearing logic is [`MpuState::copy_plan`]: it walks the parts -//! vector in order and produces a list of `UploadPartCopy` entries for the -//! gaps, coalescing adjacent gaps up to `max_merge_copy_bytes`. This is the -//! Rust port of GeeseFS's `copyUnmodifiedParts`. -//! -//! The driver function [`commit`] consumes the plan, issues backend calls, -//! and finishes with `CompleteMultipartUpload`. A small-file fast path -//! (`single_put_commit`) skips MPU entirely when no upload was ever begun. -//! -//! The "MPU started but content is sub-part" fallback (abort + materialize + -//! single PUT) is reserved for the flusher integration in a later session; -//! commits in that state currently return `Invalid` to surface the case -//! during tests. - -use std::sync::Arc; - -use bytes::Bytes; -use futures::stream::{self, StreamExt}; - -use crate::backend::{Backend, BlobMeta, CompletedPart, MultipartId, PutBlobInput}; -use crate::config::PartSchedule; -use crate::errors::{FsError, FsResult}; - -/// One uploaded part — the etag S3 hands back from `UploadPart` / -/// `UploadPartCopy`, plus a hint about how it got there (useful for tests -/// and metrics). -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct PartInfo { - pub etag: String, - pub source: PartSource, -} - -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum PartSource { - /// `UploadPart` of bytes we held in the buffer pool. - Uploaded, - /// `UploadPartCopy` from `source_offset..source_offset+source_len` of the - /// source object. Records the byte range so commit-time merges can be - /// reconstructed. - CopiedFromSelf { source_offset: u64, source_len: u64 }, -} - -/// One entry in a copy plan. Each becomes one `UploadPartCopy` call. -/// -/// `part_index` is the *new* part's slot (0-based; `part_number = part_index + 1`). -/// When a plan entry merges multiple adjacent missing slots, `part_index` is -/// the lowest slot in the merged run; the higher slots remain unfilled in -/// the parts vector and are simply omitted from `CompleteMultipartUpload`. -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct CopyPlanEntry { - pub part_index: u32, - pub source_offset: u64, - pub source_len: u64, -} - -/// Per-file MPU state. Owned by the file inode (or its handle). -#[derive(Debug)] -pub struct MpuState { - /// Destination key (where the final object will live). - pub key: String, - /// Source object's size in bytes (the value we'd `HeadObject` and read). - /// `None` means no source (fresh file). - pub source_size: Option, - /// Source object's ETag for optimistic-concurrency `If-Match` (optional; - /// not yet wired through). - pub source_etag: Option, - /// MPU id from `CreateMultipartUpload`. `None` until lazy begin. - pub upload_id: Option, - /// Per-part-index entries. `None` means "not uploaded — needs to be - /// either uploaded fresh or copied from source on commit." `parts.len()` - /// is grown as needed by `record_part_*`. - pub parts: Vec>, - /// Highest part index touched. `-1` means no parts yet. - pub high_water: i32, - /// Schedule (for `part_range` lookups during `copy_plan`). - pub schedule: PartSchedule, -} - -impl MpuState { - pub fn new( - key: String, - source_size: Option, - source_etag: Option, - schedule: PartSchedule, - ) -> Self { - Self { - key, - source_size, - source_etag, - upload_id: None, - parts: Vec::new(), - high_water: -1, - schedule, - } - } - - /// `true` iff `multipart_begin` has been called and no commit/abort yet. - pub fn has_upload(&self) -> bool { - self.upload_id.is_some() - } - - fn ensure_capacity(&mut self, part_index: u32) { - let needed = part_index as usize + 1; - if self.parts.len() < needed { - self.parts.resize(needed, None); - } - if (part_index as i32) > self.high_water { - self.high_water = part_index as i32; - } - } - - /// Record a successful `UploadPart` for `part_index`. - pub fn record_part_uploaded(&mut self, part_index: u32, etag: String) { - self.ensure_capacity(part_index); - self.parts[part_index as usize] = Some(PartInfo { - etag, - source: PartSource::Uploaded, - }); - } - - /// Record a successful `UploadPartCopy` for `part_index`. - pub fn record_part_copied( - &mut self, - part_index: u32, - source_offset: u64, - source_len: u64, - etag: String, - ) { - self.ensure_capacity(part_index); - self.parts[part_index as usize] = Some(PartInfo { - etag, - source: PartSource::CopiedFromSelf { - source_offset, - source_len, - }, - }); - } - - /// Compute the plan for `copy_unmodified_parts`. Walks part indices - /// `0..=high_water`; for each gap (`parts[i] == None`) that lies within - /// the source object, emits a `CopyPlanEntry`. Adjacent gaps are merged - /// up to `max_merge_copy_bytes`. - pub fn copy_plan(&self, max_merge_copy_bytes: u64) -> Vec { - let source_size = self.source_size.unwrap_or(0); - if source_size == 0 || self.high_water < 0 { - return Vec::new(); - } - let high = self.high_water as u32; - let mut plan = Vec::new(); - // Active merge run, if any: (start_part_index, source_start, source_end). - let mut run: Option<(u32, u64, u64)> = None; - - let flush_run = |run: &mut Option<(u32, u64, u64)>, plan: &mut Vec| { - if let Some((idx, s, e)) = run.take() { - plan.push(CopyPlanEntry { - part_index: idx, - source_offset: s, - source_len: e - s, - }); - } - }; - - for i in 0..=high { - let already_uploaded = matches!(self.parts.get(i as usize), Some(Some(_))); - let part_range = match self.schedule.part_range(i) { - Some(r) => r, - None => break, // exceeded schedule - }; - let copy_start = part_range.start; - let copy_end = part_range.end.min(source_size); - let in_source = copy_end > copy_start; - - if !already_uploaded && in_source { - let extend = match run.as_ref() { - None => false, - Some((_, run_s, run_e)) => { - let extends_contiguously = *run_e == copy_start; - let proposed_len = copy_end - *run_s; - extends_contiguously && proposed_len <= max_merge_copy_bytes - } - }; - if extend { - if let Some((_, _, run_e)) = run.as_mut() { - *run_e = copy_end; - } - } else { - flush_run(&mut run, &mut plan); - run = Some((i, copy_start, copy_end)); - } - } else { - flush_run(&mut run, &mut plan); - } - } - flush_run(&mut run, &mut plan); - plan - } - - /// Build the `CompleteMultipartUpload` payload from `parts`. Skips empty - /// slots (those covered by a merged copy) and emits the rest in - /// ascending part-number order. - pub fn build_completed_parts(&self) -> FsResult> { - let mut out = Vec::new(); - for (i, entry) in self.parts.iter().enumerate() { - if let Some(p) = entry { - out.push(CompletedPart { - part_number: (i as u32) + 1, - e_tag: p.etag.clone(), - }); - } - } - if out.is_empty() { - return Err(FsError::Invalid("commit with no parts")); - } - Ok(out) - } -} - -// --- Driver functions --------------------------------------------------- - -/// Small-file fast path: no MPU was ever begun. Issue a single `PutObject` -/// with the assembled body and return. -pub async fn single_put_commit( - backend: &dyn Backend, - key: &str, - body: Bytes, -) -> FsResult { - backend - .put_blob(PutBlobInput { - key: key.to_string(), - body, - metadata: Default::default(), - content_type: None, - }) - .await -} - -/// Drive a full MPU commit: issue every `UploadPartCopy` from the plan -/// (concurrently, capped at `max_parallel_copy`), then `CompleteMultipartUpload`. -/// -/// Preconditions: -/// - All dirty parts must have already been uploaded (caller's job). -/// - `state.upload_id` must be `Some` — call `single_put_commit` for the -/// no-MPU case. -/// - `source_key` is the key to read unchanged ranges from. For an in-place -/// update of `state.key`, pass `state.key` itself; for a fresh write that -/// never had a source, the copy plan should be empty so this argument -/// doesn't matter. -pub async fn commit( - state: &mut MpuState, - backend: Arc, - max_merge_copy_bytes: u64, - max_parallel_copy: usize, - source_key: &str, -) -> FsResult { - let upload_id = state - .upload_id - .clone() - .ok_or(FsError::Invalid("commit on MpuState without upload_id"))?; - - let plan = state.copy_plan(max_merge_copy_bytes); - if !plan.is_empty() { - let key = Arc::new(state.key.clone()); - let source_key = Arc::new(source_key.to_string()); - let upload_id = Arc::new(upload_id.clone()); - - let cap = max_parallel_copy.max(1); - let copy_results: Vec> = - stream::iter(plan.into_iter().map(|entry| { - let backend = backend.clone(); - let key = key.clone(); - let source_key = source_key.clone(); - let upload_id = upload_id.clone(); - async move { - let out = backend - .multipart_upload_part_copy( - key.as_str(), - &upload_id, - entry.part_index + 1, - source_key.as_str(), - entry.source_offset..entry.source_offset + entry.source_len, - ) - .await?; - Ok::<_, FsError>((entry, out.e_tag)) - } - })) - .buffer_unordered(cap) - .collect() - .await; - - for r in copy_results { - let (entry, etag) = r?; - state.record_part_copied( - entry.part_index, - entry.source_offset, - entry.source_len, - etag, - ); - } - } - - let parts = state.build_completed_parts()?; - let meta = backend - .multipart_complete(&state.key, &upload_id, parts) - .await?; - - // Mark complete (drop the upload_id so further operations on this state - // know the MPU is gone). - state.upload_id = None; - Ok(meta) -} - -/// Abort an in-flight MPU. Idempotent; safe to call when `upload_id` is None. -pub async fn abort(state: &mut MpuState, backend: &dyn Backend) -> FsResult<()> { - if let Some(id) = state.upload_id.take() { - backend.multipart_abort(&state.key, &id).await?; - } - Ok(()) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::backend::memory::MemoryBackend; - use crate::backend::PutBlobInput; - use crate::config::Config; - use bytes::Bytes; - use std::collections::HashMap; - use std::sync::Arc; - - fn small_schedule() -> PartSchedule { - // 4-byte parts × 100 → 400 bytes total. Easy to reason about in tests. - PartSchedule { - tiers: vec![(4, 100)], - } - } - - fn medium_schedule() -> PartSchedule { - // 5 MiB × 1000 only. Same as the first tier of the default schedule. - PartSchedule { - tiers: vec![(5 * 1024 * 1024, 1000)], - } - } - - fn fresh_backend() -> Arc { - Arc::new(MemoryBackend::new()) - } - - async fn put_source(backend: &MemoryBackend, key: &str, body: Vec) { - backend - .put_blob(PutBlobInput { - key: key.into(), - body: Bytes::from(body), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - } - - // ---- copy_plan unit tests (pure logic, no backend) ------------------- - - #[test] - fn copy_plan_no_source_is_empty() { - let s = MpuState::new("k".into(), None, None, small_schedule()); - assert!(s.copy_plan(1024).is_empty()); - } - - #[test] - fn copy_plan_no_high_water_is_empty() { - let s = MpuState::new("k".into(), Some(100), None, small_schedule()); - // No parts touched → high_water=-1 → empty. - assert!(s.copy_plan(1024).is_empty()); - } - - #[test] - fn copy_plan_all_uploaded_no_copies() { - let mut s = MpuState::new("k".into(), Some(12), None, small_schedule()); - // Source is 12 bytes = 3 full parts. Upload all 3. - s.record_part_uploaded(0, "e1".into()); - s.record_part_uploaded(1, "e2".into()); - s.record_part_uploaded(2, "e3".into()); - assert!(s.copy_plan(1024).is_empty()); - } - - #[test] - fn copy_plan_single_gap_in_middle() { - let mut s = MpuState::new("k".into(), Some(12), None, small_schedule()); - s.record_part_uploaded(0, "e1".into()); - // skip part 1 - s.record_part_uploaded(2, "e3".into()); - let plan = s.copy_plan(1024); - assert_eq!( - plan, - vec![CopyPlanEntry { - part_index: 1, - source_offset: 4, - source_len: 4 - }] - ); - } - - #[test] - fn copy_plan_merges_adjacent_gaps() { - // Source 20 bytes = 5 parts. Upload only part 0 and part 4. - // Gaps 1,2,3 should merge into one CopyPlanEntry covering bytes 4..16. - let mut s = MpuState::new("k".into(), Some(20), None, small_schedule()); - s.record_part_uploaded(0, "e0".into()); - s.record_part_uploaded(4, "e4".into()); - let plan = s.copy_plan(1024); - assert_eq!( - plan, - vec![CopyPlanEntry { - part_index: 1, - source_offset: 4, - source_len: 12 - }] - ); - } - - #[test] - fn copy_plan_respects_max_merge_bytes() { - // Same as above but with a tiny max_merge of 8 bytes. - let mut s = MpuState::new("k".into(), Some(20), None, small_schedule()); - s.record_part_uploaded(0, "e0".into()); - s.record_part_uploaded(4, "e4".into()); - let plan = s.copy_plan(8); - assert_eq!( - plan, - vec![ - CopyPlanEntry { - part_index: 1, - source_offset: 4, - source_len: 8 - }, - CopyPlanEntry { - part_index: 3, - source_offset: 12, - source_len: 4 - }, - ] - ); - } - - #[test] - fn copy_plan_clips_last_gap_to_source_size() { - // Source 10 bytes. Schedule = 4 bytes/part. high_water = 4 (as if a - // dirty write extended the file). Parts 0,4 uploaded; gaps 1,2,3. - // Source covers only bytes 0..10 → part 1 (4..8), part 2 (8..10 - // clipped from 8..12). Part 3 is entirely past source → not in plan. - let mut s = MpuState::new("k".into(), Some(10), None, small_schedule()); - s.record_part_uploaded(0, "e0".into()); - s.record_part_uploaded(4, "e4".into()); - let plan = s.copy_plan(1024); - // The merge stops where source ends. - assert_eq!( - plan, - vec![CopyPlanEntry { - part_index: 1, - source_offset: 4, - source_len: 6 // bytes 4..10 - }] - ); - } - - #[test] - fn build_completed_parts_skips_empty_slots() { - let mut s = MpuState::new("k".into(), Some(20), None, small_schedule()); - s.record_part_uploaded(0, "e0".into()); - s.record_part_copied(1, 4, 12, "ec".into()); - // slots 2,3 stay empty (they were merged into slot 1's copy). - s.record_part_uploaded(4, "e4".into()); - let parts = s.build_completed_parts().unwrap(); - let nums: Vec<_> = parts.iter().map(|p| p.part_number).collect(); - assert_eq!(nums, vec![1, 2, 5]); - } - - #[test] - fn build_completed_parts_empty_errors() { - let s = MpuState::new("k".into(), None, None, small_schedule()); - assert!(matches!( - s.build_completed_parts(), - Err(FsError::Invalid(_)) - )); - } - - // ---- single_put_commit driver ---------------------------------------- - - #[tokio::test] - async fn single_put_commit_writes_object() { - let b = fresh_backend(); - single_put_commit(b.as_ref(), "small.txt", Bytes::from_static(b"hi")) - .await - .unwrap(); - let g = b.get_blob("small.txt", None).await.unwrap(); - assert_eq!(&g.body[..], b"hi"); - } - - // ---- end-to-end MPU lifecycle via the driver ------------------------- - - #[tokio::test] - async fn mpu_lifecycle_in_place_update_uses_copy_for_unchanged_parts() { - // This is THE load-bearing test: GeeseFS-parity in-place update. - // Pre-populate "big" with 30 MiB of source bytes. - let b = fresh_backend(); - let cfg = Config::builder() - .part_schedule(medium_schedule()) - .max_merge_copy_bytes(128 * 1024 * 1024) - .build(); - let big_size: u64 = 30 * 1024 * 1024; - let mut source = vec![0u8; big_size as usize]; - // fill with a recognisable pattern - for (i, b) in source.iter_mut().enumerate() { - *b = (i % 251) as u8; - } - put_source(&b, "big", source.clone()).await; - let source_etag = b.head_blob("big").await.unwrap().e_tag; - - // Begin MPU on the same key. - let id = b - .multipart_begin(PutBlobInput { - key: "big".into(), - body: Bytes::new(), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - - let mut state = MpuState::new( - "big".into(), - Some(big_size), - Some(source_etag), - medium_schedule(), - ); - state.upload_id = Some(id.clone()); - - // Modify part_index 2 only (bytes 10MiB..15MiB). - let part_size = 5 * 1024 * 1024u64; - let mut new_part2 = vec![0u8; part_size as usize]; - new_part2.fill(0xAB); - let part2_etag = b - .multipart_upload_part( - "big", - &id, - 3, // part_number = part_index + 1 - Bytes::from(new_part2.clone()), - ) - .await - .unwrap(); - state.record_part_uploaded(2, part2_etag.e_tag); - // We need to advertise high_water = 5 (parts 0..=5 cover 30 MiB). - // Easiest way: bump high_water by recording a stub for the last index - // — but we don't want to mark it uploaded. Instead, expose a - // helper... for the test we'll just nudge directly via re-record on - // part 5 then "un-record": - state.high_water = 5; // direct manipulation in tests is fine - - // Run commit. The driver should issue UploadPartCopy for parts - // {0,1,3,4,5} merged into runs around the modified part 2. - let cfg_arc = Arc::new(cfg); - let _committed = commit( - &mut state, - b.clone() as Arc, - cfg_arc.max_merge_copy_bytes, - cfg_arc.max_parallel_copy, - "big", - ) - .await - .unwrap(); - - // Verify the resulting object: bytes [10MiB..15MiB) should be 0xAB, - // everything else should match the original source pattern. - let g = b.get_blob("big", None).await.unwrap(); - assert_eq!(g.body.len() as u64, big_size); - for i in 0..big_size as usize { - let expected = if (10 * 1024 * 1024..15 * 1024 * 1024).contains(&i) { - 0xAB - } else { - (i % 251) as u8 - }; - assert_eq!( - g.body[i], expected, - "byte {i} should be {expected:#x}, got {:#x}", - g.body[i] - ); - } - - // The MPU should be gone (no orphan). - assert_eq!(b.mpu_count(), 0); - // And the state's upload_id is cleared. - assert!(state.upload_id.is_none()); - } - - #[tokio::test] - async fn abort_clears_upload_id_idempotently() { - let b = fresh_backend(); - let mut state = MpuState::new("k".into(), None, None, small_schedule()); - // No upload_id → no-op success. - abort(&mut state, b.as_ref()).await.unwrap(); - - let id = b - .multipart_begin(PutBlobInput { - key: "k".into(), - body: Bytes::new(), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); - state.upload_id = Some(id); - assert_eq!(b.mpu_count(), 1); - abort(&mut state, b.as_ref()).await.unwrap(); - assert_eq!(b.mpu_count(), 0); - assert!(state.upload_id.is_none()); - // Idempotent. - abort(&mut state, b.as_ref()).await.unwrap(); - } - - #[tokio::test] - async fn commit_without_upload_id_errors() { - let b = fresh_backend(); - let mut state = MpuState::new("k".into(), None, None, small_schedule()); - let r = commit(&mut state, b.clone() as Arc, 1024, 4, "k").await; - assert!(matches!(r, Err(FsError::Invalid(_)))); - } -} diff --git a/crates/s3fs-core/src/path.rs b/crates/s3fs-core/src/path.rs index 2050778..bd7837a 100644 --- a/crates/s3fs-core/src/path.rs +++ b/crates/s3fs-core/src/path.rs @@ -307,8 +307,8 @@ mod tests { fn rejects_total_key_too_long() { // Build a path where each segment is valid but the total exceeds 1024. let seg = "a".repeat(MAX_SEGMENT_LEN); // 255 - let path = std::iter::repeat(seg.as_str()) - .take(5) // 5*255 + 4 separators = 1279 > 1024 + // 5*255 + 4 separators = 1279 > 1024 + let path = std::iter::repeat_n(seg.as_str(), 5) .collect::>() .join("/"); assert!(matches!( diff --git a/crates/s3fs-core/src/rename.rs b/crates/s3fs-core/src/rename.rs deleted file mode 100644 index 4beb443..0000000 --- a/crates/s3fs-core/src/rename.rs +++ /dev/null @@ -1,230 +0,0 @@ -//! Async rename queue. -//! -//! `Fs::rename` rewires the inode tree synchronously and returns -//! immediately; the underlying S3 work (CopyObject + DeleteObject for a -//! file, paginated recursion for a directory) runs on a long-lived -//! background worker. -//! -//! While the worker is in flight, the destination inode carries a -//! `RenameState { old_key, … }`; `InodeTree::current_s3_key` returns the -//! OLD key for any read/write so callers continue to see the existing -//! object. Once the worker finishes, `rename_state` is cleared and the -//! natural new key is used. -//! -//! Errors from the worker are stashed on `RenameState::error` and surface -//! on the next `Fs::sync` / `Fs::rename` for that inode. - -use std::sync::Arc; - -use tokio::sync::mpsc; -use tokio_util::sync::CancellationToken; - -use crate::backend::{Backend, CopyBlobInput, ListBlobsInput}; -use crate::errors::{FsError, FsResult}; -use crate::inode::Inode; - -/// One unit of work for the rename worker. -pub enum RenameJob { - File { - backend: Arc, - inode: Arc, - old_key: String, - new_key: String, - }, - Dir { - backend: Arc, - inode: Arc, - old_prefix: String, - new_prefix: String, - }, -} - -impl std::fmt::Debug for RenameJob { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - RenameJob::File { - old_key, new_key, .. - } => write!(f, "RenameJob::File {{ {old_key} -> {new_key} }}"), - RenameJob::Dir { - old_prefix, - new_prefix, - .. - } => write!(f, "RenameJob::Dir {{ {old_prefix} -> {new_prefix} }}"), - } - } -} - -#[derive(Debug)] -pub struct RenameQueue { - job_tx: mpsc::UnboundedSender, - cancel: CancellationToken, - worker: parking_lot::Mutex>>, -} - -impl RenameQueue { - pub fn new() -> Self { - let (job_tx, mut job_rx) = mpsc::unbounded_channel::(); - let cancel = CancellationToken::new(); - let cancel_for_worker = cancel.clone(); - let worker = tokio::spawn(async move { - loop { - tokio::select! { - biased; - _ = cancel_for_worker.cancelled() => break, - maybe_job = job_rx.recv() => { - let Some(job) = maybe_job else { break }; - // Each job runs in its own task so a slow recursive - // dir rename doesn't head-of-line block file renames. - tokio::spawn(run_job(job)); - } - } - } - }); - Self { - job_tx, - cancel, - worker: parking_lot::Mutex::new(Some(worker)), - } - } - - pub fn enqueue(&self, job: RenameJob) -> FsResult<()> { - self.job_tx - .send(job) - .map_err(|_| FsError::Io("rename queue closed".into())) - } -} - -impl Default for RenameQueue { - fn default() -> Self { - Self::new() - } -} - -impl Drop for RenameQueue { - fn drop(&mut self) { - self.cancel.cancel(); - if let Some(h) = self.worker.lock().take() { - h.abort(); - } - } -} - -async fn run_job(job: RenameJob) { - match job { - RenameJob::File { - backend, - inode, - old_key, - new_key, - } => { - let res = run_file_rename(backend, &inode, &old_key, &new_key).await; - finish(&inode, res); - } - RenameJob::Dir { - backend, - inode, - old_prefix, - new_prefix, - } => { - let res = run_dir_rename(backend, &inode, &old_prefix, &new_prefix).await; - finish(&inode, res); - } - } -} - -async fn run_file_rename( - backend: Arc, - inode: &Arc, - old_key: &str, - new_key: &str, -) -> FsResult<()> { - let _g = inode.rename_lock.lock().await; - backend - .copy_blob(CopyBlobInput { - source_key: old_key.to_string(), - destination_key: new_key.to_string(), - replace_metadata: None, - replace_content_type: None, - }) - .await?; - backend.delete_blob(old_key).await?; - Ok(()) -} - -async fn run_dir_rename( - backend: Arc, - inode: &Arc, - old_prefix: &str, - new_prefix: &str, -) -> FsResult<()> { - let _g = inode.rename_lock.lock().await; - let mut continuation: Option = None; - loop { - let listing = backend - .list_blobs(ListBlobsInput { - prefix: old_prefix, - delimiter: None, - continuation_token: continuation.as_deref(), - max_keys: None, - ..Default::default() - }) - .await?; - for item in &listing.items { - let suffix = match item.key.strip_prefix(old_prefix) { - Some(s) => s, - None => continue, - }; - let dst_key = format!("{new_prefix}{suffix}"); - backend - .copy_blob(CopyBlobInput { - source_key: item.key.clone(), - destination_key: dst_key, - replace_metadata: None, - replace_content_type: None, - }) - .await?; - backend.delete_blob(&item.key).await?; - } - if !listing.is_truncated { - break; - } - match listing.next_continuation_token { - Some(t) => continuation = Some(t), - None => break, - } - } - Ok(()) -} - -/// Clear the inode's `rename_state` on success (so future ops use the new -/// key); on error, stash the error in `rename_state.error` so the next -/// `sync` for the inode surfaces it. Always notify `done` so any waiter -/// (e.g. tests, future explicit `wait_for_rename`) wakes up. -fn finish(inode: &Arc, res: FsResult<()>) { - let state = inode.rename_state(); - match res { - Ok(()) => { - *inode.rename_state.write() = None; - if let Some(s) = state { - s.done.notify_waiters(); - } - } - Err(e) => { - if let Some(s) = state { - *s.error.write() = Some(e); - s.done.notify_waiters(); - } - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn drop_aborts_worker_cleanly() { - let q = RenameQueue::new(); - drop(q); // should not panic / hang - } -} diff --git a/crates/s3fs-core/src/store/blkptr.rs b/crates/s3fs-core/src/store/blkptr.rs new file mode 100644 index 0000000..0b61574 --- /dev/null +++ b/crates/s3fs-core/src/store/blkptr.rs @@ -0,0 +1,445 @@ +//! `BlkPtr` — the block pointer, and the edge of the Merkle tree. +//! +//! A block pointer says *where* a block lives, *how* to decrypt it, and *what +//! it should hash to*. Because a parent block is itself a block, and its own +//! pointer carries its own checksum, the checksums chain all the way to the +//! root record. Verifying the root therefore verifies every byte beneath it. +//! +//! ## Location, not content +//! +//! The address is a [`Dva`] — a `(txg, slab, offset, len)` tuple naming a byte +//! range inside an immutable slab object — rather than a content hash. This is +//! the ZFS model, and it is chosen deliberately over content-addressed +//! `blocks/` keys: every block dirtied in a transaction group is packed +//! into a handful of slabs, so a commit costs a couple of PUTs no matter how +//! many blocks changed. One PUT per 128 KiB block would make small-file +//! workloads (a SQLite page write, say) cost a round trip each. +//! +//! ## Wire layout — 128 bytes, big-endian +//! +//! ```text +//! offset size field +//! 0 8 dva.txg transaction group that allocated the slab +//! 8 2 dva.slab slab index within that txg +//! 10 4 dva.offset byte offset within the slab +//! 14 4 dva.len stored length (ciphertext + AEAD tag) +//! 18 4 logical_len plaintext length +//! 22 1 level 0 = data, >=1 = indirect +//! 23 1 flags low nibble: compression algorithm +//! 24 4 fill non-hole pointers beneath this one +//! 28 8 birth_txg txg that wrote this block +//! 36 12 nonce AEAD nonce (txg || block_seq) +//! 48 32 checksum BLAKE3-256 of the stored bytes +//! 80 48 reserved must be zero +//! ``` +//! +//! Big-endian throughout, matching the AAD and nonce encodings, so a hex dump +//! of a slab reads in the same order as the struct. +//! +//! An all-zero pointer is a *hole*: a sparse region that reads as zeros and +//! occupies no storage. Holes are why growing a file with `set_size` is free. + +use crate::crypto::aead::TAG_LEN; +use crate::crypto::{BlockNonce, Hash256}; +use crate::errors::{FsError, FsResult}; + +/// Encoded size of a block pointer. +pub const BLKPTR_LEN: usize = 128; + +/// Offset of the `reserved` tail within the encoding. +const RESERVED_OFFSET: usize = 80; + +/// Deepest indirect tree we will decode. +/// +/// The binding case is the *smallest* record size, not the default. At 128 KiB +/// the fan-out is 1024 and six levels already address more than any real file; +/// but at the 4 KiB minimum the fan-out is only 32, and a file spanning the +/// full `u64` byte range needs 2^52 blocks, hence `1 + ceil(52 / 5) = 12` +/// levels. The limit is set from that worst case with one level to spare. +/// +/// Anything deeper is a corrupt or hostile pointer. Rejecting it at decode +/// bounds recursion in the tree walker. +pub const MAX_LEVEL: u8 = 12; + +/// Largest block we will decode, as a sanity bound on `logical_len`. +/// +/// Without this, a corrupt length field turns into a multi-gigabyte allocation +/// on the read path — a trivial denial of service from anyone who can write to +/// the bucket. +pub const MAX_BLOCK_LEN: u32 = 16 * 1024 * 1024; + +/// How many block pointers fit in one block of the given size. +pub const fn blkptrs_per_block(block_size: usize) -> usize { + block_size / BLKPTR_LEN +} + +/// Compression applied before encryption. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +#[repr(u8)] +pub enum Compression { + #[default] + None = 0, +} + +impl Compression { + fn from_u8(v: u8) -> FsResult { + match v { + 0 => Ok(Compression::None), + _ => Err(FsError::Integrity("blkptr: unknown compression algorithm")), + } + } +} + +/// Data Virtual Address — a byte range inside an immutable slab object. +/// +/// Hashable because it is the block cache's key: a DVA is allocated once and +/// never reused, so it identifies an immutable byte range for all time. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Default)] +pub struct Dva { + /// Transaction group whose slab holds this block. + pub txg: u64, + /// Slab index within that txg. + pub slab: u16, + /// Byte offset within the slab. + pub offset: u32, + /// Stored length: ciphertext plus AEAD tag. + pub len: u32, +} + +impl Dva { + /// Byte range within the slab, for a ranged GET. + pub fn range(&self) -> std::ops::Range { + let start = u64::from(self.offset); + start..start + u64::from(self.len) + } +} + +/// A pointer to one block. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub struct BlkPtr { + pub dva: Dva, + /// Plaintext length. Differs from `dva.len` by the AEAD tag and, once + /// compression is implemented, by the compression ratio. + pub logical_len: u32, + /// 0 for data blocks, ≥1 for indirect blocks. + pub level: u8, + pub compression: Compression, + /// Number of non-hole pointers at or beneath this one. Lets a sparse file + /// report its allocated size, and lets a tree walk skip empty subtrees + /// without reading them. + pub fill: u32, + /// Transaction group that wrote this block. Part of the AEAD AAD, so it + /// cannot be altered without invalidating the block. + pub birth_txg: u64, + pub nonce: BlockNonce, + /// BLAKE3-256 of the stored (encrypted) bytes. + pub checksum: Hash256, +} + +impl BlkPtr { + /// The all-zero pointer: a sparse hole. + pub const HOLE: BlkPtr = BlkPtr { + dva: Dva { + txg: 0, + slab: 0, + offset: 0, + len: 0, + }, + logical_len: 0, + level: 0, + compression: Compression::None, + fill: 0, + birth_txg: 0, + nonce: BlockNonce::from_bytes([0u8; 12]), + checksum: Hash256::ZERO, + }; + + /// `true` if this pointer addresses nothing — a sparse region that reads + /// as zeros. + pub fn is_hole(&self) -> bool { + self.dva.len == 0 && self.checksum.is_zero() + } + + pub fn encode(&self) -> [u8; BLKPTR_LEN] { + let mut b = [0u8; BLKPTR_LEN]; + b[0..8].copy_from_slice(&self.dva.txg.to_be_bytes()); + b[8..10].copy_from_slice(&self.dva.slab.to_be_bytes()); + b[10..14].copy_from_slice(&self.dva.offset.to_be_bytes()); + b[14..18].copy_from_slice(&self.dva.len.to_be_bytes()); + b[18..22].copy_from_slice(&self.logical_len.to_be_bytes()); + b[22] = self.level; + b[23] = self.compression as u8; + b[24..28].copy_from_slice(&self.fill.to_be_bytes()); + b[28..36].copy_from_slice(&self.birth_txg.to_be_bytes()); + b[36..48].copy_from_slice(self.nonce.as_bytes()); + b[48..80].copy_from_slice(self.checksum.as_bytes()); + // b[80..128] stays zero. + b + } + + /// Decode and validate. + /// + /// Validation is strict — every rejection here is a pointer that could + /// only have come from corruption or tampering, and the cost of accepting + /// one is either wrong data or an unbounded allocation. Unknown bits are + /// rejected rather than ignored; the root record's `format_version` is the + /// mechanism for evolving this layout. + pub fn decode(bytes: &[u8]) -> FsResult { + if bytes.len() != BLKPTR_LEN { + return Err(FsError::Integrity("blkptr: wrong length")); + } + if bytes[RESERVED_OFFSET..].iter().any(|&b| b != 0) { + return Err(FsError::Integrity("blkptr: reserved bytes not zero")); + } + if bytes.iter().all(|&b| b == 0) { + return Ok(BlkPtr::HOLE); + } + + let ptr = BlkPtr { + dva: Dva { + txg: u64::from_be_bytes(bytes[0..8].try_into().expect("8 bytes")), + slab: u16::from_be_bytes(bytes[8..10].try_into().expect("2 bytes")), + offset: u32::from_be_bytes(bytes[10..14].try_into().expect("4 bytes")), + len: u32::from_be_bytes(bytes[14..18].try_into().expect("4 bytes")), + }, + logical_len: u32::from_be_bytes(bytes[18..22].try_into().expect("4 bytes")), + level: bytes[22], + compression: Compression::from_u8(bytes[23])?, + fill: u32::from_be_bytes(bytes[24..28].try_into().expect("4 bytes")), + birth_txg: u64::from_be_bytes(bytes[28..36].try_into().expect("8 bytes")), + nonce: BlockNonce::from_bytes(bytes[36..48].try_into().expect("12 bytes")), + checksum: Hash256::from_bytes(bytes[48..80].try_into().expect("32 bytes")), + }; + + // A non-hole must actually point somewhere. Catching a partially-zero + // pointer here stops it being mistaken for a hole, which would silently + // turn tampering into a file full of zeros. + if ptr.checksum.is_zero() { + return Err(FsError::Integrity("blkptr: non-hole with zero checksum")); + } + if ptr.dva.len < TAG_LEN as u32 { + return Err(FsError::Integrity("blkptr: stored length below AEAD tag")); + } + if ptr.level > MAX_LEVEL { + return Err(FsError::Integrity("blkptr: level exceeds maximum")); + } + if ptr.logical_len > MAX_BLOCK_LEN || ptr.dva.len > MAX_BLOCK_LEN { + return Err(FsError::Integrity("blkptr: block length exceeds maximum")); + } + // Uncompressed blocks are exactly plaintext plus tag. This catches a + // length field edited to over- or under-read the slab. + if ptr.compression == Compression::None && ptr.dva.len != ptr.logical_len + TAG_LEN as u32 { + return Err(FsError::Integrity( + "blkptr: stored length inconsistent with logical length", + )); + } + if ptr.dva.offset.checked_add(ptr.dva.len).is_none() { + return Err(FsError::Integrity("blkptr: slab range overflows")); + } + Ok(ptr) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn sample() -> BlkPtr { + BlkPtr { + dva: Dva { + txg: 0x0102_0304_0506_0708, + slab: 0x090a, + offset: 0x0b0c_0d0e, + len: 4096 + TAG_LEN as u32, + }, + logical_len: 4096, + level: 2, + compression: Compression::None, + fill: 7, + birth_txg: 0x1112_1314_1516_1718, + nonce: BlockNonce::new(0x1112_1314_1516_1718, 5), + checksum: Hash256::of(b"block bytes"), + } + } + + #[test] + fn encoding_is_exactly_128_bytes() { + assert_eq!(sample().encode().len(), BLKPTR_LEN); + assert_eq!(BLKPTR_LEN, 128); + } + + #[test] + fn round_trips() { + let p = sample(); + assert_eq!(BlkPtr::decode(&p.encode()).unwrap(), p); + } + + #[test] + fn hole_round_trips_and_is_all_zero() { + let encoded = BlkPtr::HOLE.encode(); + assert!(encoded.iter().all(|&b| b == 0)); + assert_eq!(BlkPtr::decode(&encoded).unwrap(), BlkPtr::HOLE); + assert!(BlkPtr::HOLE.is_hole()); + assert!(!sample().is_hole()); + } + + #[test] + fn field_offsets_are_stable() { + // The on-disk layout is a compatibility surface: if these move, + // existing filesystems become unreadable. Pin them explicitly rather + // than trusting the round-trip test, which would pass even if two + // fields swapped places. + let b = sample().encode(); + assert_eq!(&b[0..8], &0x0102_0304_0506_0708u64.to_be_bytes()); + assert_eq!(&b[8..10], &0x090au16.to_be_bytes()); + assert_eq!(&b[10..14], &0x0b0c_0d0eu32.to_be_bytes()); + assert_eq!(&b[14..18], &(4096u32 + TAG_LEN as u32).to_be_bytes()); + assert_eq!(&b[18..22], &4096u32.to_be_bytes()); + assert_eq!(b[22], 2); + assert_eq!(b[23], 0); + assert_eq!(&b[24..28], &7u32.to_be_bytes()); + assert_eq!(&b[28..36], &0x1112_1314_1516_1718u64.to_be_bytes()); + assert_eq!( + &b[36..48], + BlockNonce::new(0x1112_1314_1516_1718, 5).as_bytes() + ); + assert_eq!(&b[48..80], Hash256::of(b"block bytes").as_bytes()); + assert!(b[80..].iter().all(|&x| x == 0)); + } + + #[test] + fn rejects_wrong_length() { + assert!(BlkPtr::decode(&[0u8; 127]).is_err()); + assert!(BlkPtr::decode(&[0u8; 129]).is_err()); + assert!(BlkPtr::decode(&[]).is_err()); + } + + #[test] + fn rejects_nonzero_reserved_bytes() { + let mut b = sample().encode(); + b[100] = 1; + assert!(matches!( + BlkPtr::decode(&b), + Err(FsError::Integrity("blkptr: reserved bytes not zero")) + )); + } + + /// The dangerous near-miss: a pointer zeroed everywhere that matters would + /// read as a hole and silently produce a file of zeros. It must be an + /// error instead. + #[test] + fn rejects_partially_zeroed_pointer() { + let mut b = sample().encode(); + b[48..80].fill(0); // wipe the checksum, leave the address + assert!(matches!( + BlkPtr::decode(&b), + Err(FsError::Integrity("blkptr: non-hole with zero checksum")) + )); + } + + #[test] + fn rejects_stored_length_below_tag() { + let mut p = sample(); + p.dva.len = TAG_LEN as u32 - 1; + p.logical_len = 0; + assert!(matches!( + BlkPtr::decode(&p.encode()), + Err(FsError::Integrity("blkptr: stored length below AEAD tag")) + )); + } + + #[test] + fn rejects_length_inconsistency() { + // Over-read: claim more stored bytes than the plaintext accounts for. + let mut p = sample(); + p.dva.len += 1; + assert!(BlkPtr::decode(&p.encode()).is_err()); + + // Under-read. + let mut p = sample(); + p.logical_len += 1; + assert!(BlkPtr::decode(&p.encode()).is_err()); + } + + #[test] + fn rejects_excessive_level() { + let mut p = sample(); + p.level = MAX_LEVEL + 1; + assert!(matches!( + BlkPtr::decode(&p.encode()), + Err(FsError::Integrity("blkptr: level exceeds maximum")) + )); + + p.level = MAX_LEVEL; + assert!(BlkPtr::decode(&p.encode()).is_ok()); + } + + #[test] + fn rejects_absurd_block_length() { + let mut p = sample(); + p.logical_len = MAX_BLOCK_LEN + 1; + p.dva.len = p.logical_len + TAG_LEN as u32; + assert!(matches!( + BlkPtr::decode(&p.encode()), + Err(FsError::Integrity("blkptr: block length exceeds maximum")) + )); + } + + #[test] + fn rejects_unknown_compression() { + let mut b = sample().encode(); + b[23] = 9; + assert!(matches!( + BlkPtr::decode(&b), + Err(FsError::Integrity("blkptr: unknown compression algorithm")) + )); + } + + #[test] + fn rejects_slab_range_overflow() { + let mut p = sample(); + p.dva.offset = u32::MAX; + assert!(matches!( + BlkPtr::decode(&p.encode()), + Err(FsError::Integrity("blkptr: slab range overflows")) + )); + } + + #[test] + fn dva_range_covers_the_stored_bytes() { + let p = sample(); + assert_eq!( + p.dva.range(), + 0x0b0c_0d0e..0x0b0c_0d0e + 4096 + TAG_LEN as u64 + ); + } + + #[test] + fn fanout_at_default_record_size() { + assert_eq!(blkptrs_per_block(128 * 1024), 1024); + assert_eq!(blkptrs_per_block(4096), 32); + } + + /// Every single-byte corruption of a valid pointer must either decode to + /// something different (and so fail its checksum against the parent) or be + /// rejected outright. What must never happen is decoding back to the + /// original pointer, which would mean a byte of the encoding is ignored. + #[test] + fn no_byte_of_the_encoding_is_ignored() { + let p = sample(); + let encoded = p.encode(); + for i in 0..BLKPTR_LEN { + for bit in [0x01u8, 0x80] { + let mut b = encoded; + b[i] ^= bit; + match BlkPtr::decode(&b) { + Err(_) => {} + Ok(decoded) => assert_ne!( + decoded, p, + "flipping bit {bit:#x} of byte {i} decoded back to the original" + ), + } + } + } + } +} diff --git a/crates/s3fs-core/src/store/blockstore.rs b/crates/s3fs-core/src/store/blockstore.rs new file mode 100644 index 0000000..e82fd7f --- /dev/null +++ b/crates/s3fs-core/src/store/blockstore.rs @@ -0,0 +1,358 @@ +//! `BlockStore` — the verified read path and the slab-writing commit path. +//! +//! Everything above this module deals in `BlkPtr`s and plaintext blocks and +//! never touches the backend directly. Everything below is ranged GETs and +//! PUTs against immutable objects. +//! +//! The read path is fail-closed by construction: there is no way to obtain a +//! block's bytes except through [`BlockStore::read_block`], which verifies the +//! BLAKE3 checksum against the parent's pointer and then AEAD-opens with +//! position-binding AAD. A caller cannot opt out, and no partial result +//! escapes on failure. + +use std::sync::Arc; + +use bytes::Bytes; +use futures::stream::{FuturesUnordered, StreamExt}; + +use crate::backend::{Backend, ObjectLock, PutBlobInput}; +use crate::crypto::KeyMaterial; +use crate::errors::{FsError, FsResult}; + +use super::blkptr::BlkPtr; +use super::cache::{BlockCache, BlockKey, CacheStats}; +use super::config::StoreConfig; +use super::slab::{verify_and_open, FinishedSlab}; + +/// Reads and writes verified, encrypted blocks against a backend. +#[derive(Debug)] +pub struct BlockStore { + data: Arc, + keys: Arc, + config: Arc, + cache: BlockCache, +} + +impl BlockStore { + pub fn new(data: Arc, keys: Arc, config: Arc) -> Self { + let cache = BlockCache::new(config.block_cache_bytes); + Self { + data, + keys, + config, + cache, + } + } + + pub fn config(&self) -> &StoreConfig { + &self.config + } + + pub fn keys(&self) -> &KeyMaterial { + &self.keys + } + + pub fn cache_stats(&self) -> CacheStats { + self.cache.stats() + } + + /// Read, verify, and decrypt the block a pointer addresses. + /// + /// A hole yields `logical_len` zero bytes without any I/O — that is what + /// makes sparse files free. + /// + /// `objid` and `block_index` must be the position the caller *expects* the + /// block to occupy, not a position read back from the store. They are fed + /// into the AEAD's additional data, so passing what the store told us + /// would defeat the check entirely. + pub async fn read_block(&self, ptr: &BlkPtr, objid: u64, block_index: u64) -> FsResult { + if ptr.is_hole() { + return Ok(Bytes::from(vec![0u8; ptr.logical_len as usize])); + } + let cache_key = BlockKey::new(ptr.dva, objid, block_index); + if let Some(hit) = self.cache.get(&cache_key) { + return Ok(hit); + } + + let key = self.config.slab_key(ptr.dva.txg, ptr.dva.slab); + let got = self + .data + .get_blob(&key, Some(ptr.dva.range())) + .await + .map_err(|e| match e { + // A missing slab under a root that references it means the + // data was deleted out from under us. That is a denial of + // service, not a forgery — but it is still fatal for this read + // and must not be confused with "the file does not exist". + FsError::NotFound => FsError::Integrity("slab referenced by root is missing"), + other => other, + })?; + + let plaintext = Bytes::from(verify_and_open( + &self.keys, + ptr, + objid, + block_index, + &got.body, + )?); + self.cache.insert(cache_key, plaintext.clone()); + Ok(plaintext) + } + + /// PUT every slab of a transaction group, in parallel. + /// + /// Returns only once all of them are durable. The commit protocol depends + /// on this: the root record must not be published until every block it + /// references is readable, or a crash between the two would leave a root + /// pointing at blocks that do not exist. + pub async fn write_slabs(&self, txg: u64, slabs: Vec) -> FsResult<()> { + if slabs.is_empty() { + return Ok(()); + } + let limit = self.config.max_parallel_slab_puts; + let mut in_flight = FuturesUnordered::new(); + let mut queue = slabs.into_iter(); + + loop { + while in_flight.len() < limit { + match queue.next() { + Some(slab) => in_flight.push(self.put_one_slab(txg, slab)), + None => break, + } + } + match in_flight.next().await { + Some(result) => result?, + None => return Ok(()), + } + } + } + + async fn put_one_slab(&self, txg: u64, slab: FinishedSlab) -> FsResult<()> { + let key = self.config.slab_key(txg, slab.index); + // Slabs carry no retention: they live in the unlocked data bucket, so + // dead copy-on-write blocks stay reclaimable. Deleting one is a + // detectable denial of service, never a rollback — the rollback + // guarantee lives entirely in the locked root records. + self.data + .put_blob(PutBlobInput::new(key, slab.body)) + .await + .map(|_| ()) + } + + /// Retention to stamp on a root record, resolved against wall-clock now. + pub fn root_retention(&self) -> Option { + self.config.root_retention.map(|d| ObjectLock { + mode: crate::backend::ObjectLockMode::Compliance, + retain_until: std::time::SystemTime::now() + d, + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::backend::memory::MemoryBackend; + use crate::crypto::MasterSecret; + use crate::store::slab::SlabWriter; + + fn store(config: StoreConfig) -> (Arc, BlockStore) { + let backend = Arc::new(MemoryBackend::new()); + let keys = + Arc::new(KeyMaterial::derive(&MasterSecret::from_bytes([4u8; 32]), [0u8; 16]).unwrap()); + let bs = BlockStore::new(backend.clone(), keys, Arc::new(config)); + (backend, bs) + } + + /// Seal `blocks` into slabs, PUT them, and return their pointers. + async fn commit_blocks( + bs: &BlockStore, + txg: u64, + blocks: &[(u64, u64, Vec)], + ) -> Vec { + let mut w = SlabWriter::new(txg, bs.config()); + let ptrs: Vec<_> = blocks + .iter() + .map(|(objid, idx, data)| w.write_block(bs.keys(), *objid, 0, *idx, 1, data).unwrap()) + .collect(); + bs.write_slabs(txg, w.finish().unwrap()).await.unwrap(); + ptrs + } + + #[tokio::test] + async fn round_trip_through_the_backend() { + let (_b, bs) = store(StoreConfig::default()); + let data = vec![0x5au8; 8192]; + let ptrs = commit_blocks(&bs, 1, &[(10, 0, data.clone())]).await; + + assert_eq!(bs.read_block(&ptrs[0], 10, 0).await.unwrap(), data); + } + + #[tokio::test] + async fn holes_read_as_zeros_without_io() { + let (backend, bs) = store(StoreConfig::default()); + let mut hole = BlkPtr::HOLE; + hole.logical_len = 4096; + + let got = bs.read_block(&hole, 1, 0).await.unwrap(); + assert_eq!(got.len(), 4096); + assert!(got.iter().all(|&b| b == 0)); + assert_eq!(backend.object_count(), 0, "a hole must not touch storage"); + } + + #[tokio::test] + async fn second_read_is_served_from_cache() { + let (_b, bs) = store(StoreConfig::default()); + let ptrs = commit_blocks(&bs, 1, &[(10, 0, vec![1u8; 4096])]).await; + + bs.read_block(&ptrs[0], 10, 0).await.unwrap(); + bs.read_block(&ptrs[0], 10, 0).await.unwrap(); + let stats = bs.cache_stats(); + assert_eq!((stats.hits, stats.misses), (1, 1)); + } + + #[tokio::test] + async fn many_blocks_pack_into_few_objects() { + // The economic claim behind slab packing: a txg dirtying many blocks + // costs a handful of PUTs, not one per block. + let (backend, bs) = store(StoreConfig::default()); + let blocks: Vec<_> = (0..64u64).map(|i| (7, i, vec![i as u8; 4096])).collect(); + let ptrs = commit_blocks(&bs, 3, &blocks).await; + + assert_eq!(ptrs.len(), 64); + assert_eq!(backend.object_count(), 1, "64 blocks should be one slab"); + + for (i, p) in ptrs.iter().enumerate() { + assert_eq!( + bs.read_block(p, 7, i as u64).await.unwrap(), + vec![i as u8; 4096] + ); + } + } + + #[tokio::test] + async fn multiple_slabs_are_all_written() { + let cfg = StoreConfig { + record_size: 4096, + slab_max_bytes: 8300, + ..Default::default() + }; + let (backend, bs) = store(cfg); + let blocks: Vec<_> = (0..7u64).map(|i| (1, i, vec![i as u8; 4096])).collect(); + let ptrs = commit_blocks(&bs, 1, &blocks).await; + + assert_eq!(backend.object_count(), 4, "7 blocks, 2 per slab"); + for (i, p) in ptrs.iter().enumerate() { + assert_eq!( + bs.read_block(p, 1, i as u64).await.unwrap(), + vec![i as u8; 4096] + ); + } + } + + #[tokio::test] + async fn empty_commit_writes_nothing() { + let (backend, bs) = store(StoreConfig::default()); + bs.write_slabs(1, vec![]).await.unwrap(); + assert_eq!(backend.object_count(), 0); + } + + // ---- adversarial reads ------------------------------------------------- + + #[tokio::test] + async fn tampered_slab_bytes_are_rejected() { + let (backend, bs) = store(StoreConfig::default()); + let ptrs = commit_blocks(&bs, 1, &[(10, 0, vec![0xaau8; 4096])]).await; + + // Rewrite the slab with corrupted contents, as a bucket operator could. + let key = bs.config().slab_key(1, 0); + let mut body = backend.get_blob(&key, None).await.unwrap().body.to_vec(); + body[0] ^= 0xff; + backend + .put_blob(PutBlobInput::new(key, Bytes::from(body))) + .await + .unwrap(); + + assert!(matches!( + bs.read_block(&ptrs[0], 10, 0).await, + Err(FsError::Integrity(_)) + )); + } + + #[tokio::test] + async fn deleted_slab_reports_integrity_not_missing_file() { + let (backend, bs) = store(StoreConfig::default()); + let ptrs = commit_blocks(&bs, 1, &[(10, 0, vec![1u8; 4096])]).await; + backend + .delete_blob(&bs.config().slab_key(1, 0)) + .await + .unwrap(); + + // Must not look like "no such file" — the root says this block exists, + // so its absence is the store failing us, not the guest asking for + // something that was never there. + assert!(matches!( + bs.read_block(&ptrs[0], 10, 0).await, + Err(FsError::Integrity("slab referenced by root is missing")) + )); + } + + #[tokio::test] + async fn a_block_cannot_be_read_at_the_wrong_position() { + let (_b, bs) = store(StoreConfig::default()); + let ptrs = commit_blocks(&bs, 1, &[(10, 5, vec![9u8; 1024])]).await; + + assert!(bs.read_block(&ptrs[0], 10, 5).await.is_ok()); + assert!(bs.read_block(&ptrs[0], 11, 5).await.is_err(), "wrong objid"); + assert!(bs.read_block(&ptrs[0], 10, 6).await.is_err(), "wrong index"); + } + + /// Point a pointer at a neighbouring block's bytes. Both blocks are + /// genuine; only the checksum in the pointer disagrees. + #[tokio::test] + async fn a_pointer_cannot_be_aimed_at_another_block() { + let (_b, bs) = store(StoreConfig::default()); + let ptrs = commit_blocks( + &bs, + 1, + &[(10, 0, vec![1u8; 1024]), (10, 1, vec![2u8; 1024])], + ) + .await; + + let mut forged = ptrs[0]; + forged.dva.offset = ptrs[1].dva.offset; + assert!(matches!( + bs.read_block(&forged, 10, 0).await, + Err(FsError::Integrity("block: checksum mismatch")) + )); + } + + #[tokio::test] + async fn a_block_from_another_filesystem_is_rejected() { + let (backend, bs) = store(StoreConfig::default()); + let ptrs = commit_blocks(&bs, 1, &[(10, 0, vec![1u8; 1024])]).await; + + // Same bucket, same pointers, different key material. + let other_keys = Arc::new( + KeyMaterial::derive(&MasterSecret::from_bytes([99u8; 32]), [0u8; 16]).unwrap(), + ); + let other = BlockStore::new(backend, other_keys, Arc::new(StoreConfig::default())); + assert!(matches!( + other.read_block(&ptrs[0], 10, 0).await, + Err(FsError::Integrity(_)) + )); + } + + #[tokio::test] + async fn root_retention_is_compliance_mode_and_in_the_future() { + let (_b, bs) = store(StoreConfig::default()); + let lock = bs.root_retention().expect("default config locks roots"); + assert_eq!(lock.mode, crate::backend::ObjectLockMode::Compliance); + assert!(lock.retain_until > std::time::SystemTime::now()); + + let (_b, unlocked) = store(StoreConfig { + root_retention: None, + ..Default::default() + }); + assert!(unlocked.root_retention().is_none()); + } +} diff --git a/crates/s3fs-core/src/store/cache.rs b/crates/s3fs-core/src/store/cache.rs new file mode 100644 index 0000000..c49ba2e --- /dev/null +++ b/crates/s3fs-core/src/store/cache.rs @@ -0,0 +1,289 @@ +//! Byte-accounted LRU cache of decrypted blocks. +//! +//! Entries never go stale and need no invalidation. Copy-on-write gives us +//! that for free: a modified block is written to a *new* address, so a cached +//! entry describes bytes that can never change, and the old entry simply ages +//! out. +//! +//! Entries are stored decrypted, so a hit skips the GET, the BLAKE3 pass, and +//! the AEAD open. That is safe because the cache lives in enclave memory, +//! which is the same trust boundary the plaintext already occupies. +//! +//! ## Why the key includes the position +//! +//! A hit bypasses verification, so the key must carry everything verification +//! would have checked. The address alone is not enough: keyed only by +//! [`Dva`], a block legitimately read at one position would then be served +//! from cache when a *forged* indirect block claimed it belonged at another — +//! exactly the relocation attack the AEAD's additional data exists to stop. +//! Including `(objid, block_index)` in the key turns that into a miss, and the +//! miss then fails the real check. +//! +//! No legitimate tree ever holds one address at two positions: a block sealed +//! for one position cannot be opened at another, so the duplicate keys this +//! permits in principle do not arise in practice. + +use bytes::Bytes; +use hashlink::LinkedHashMap; +use parking_lot::Mutex; + +use super::blkptr::Dva; + +/// Cache key: an immutable address plus the tree position it was verified at. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub struct BlockKey { + pub dva: Dva, + pub objid: u64, + pub block_index: u64, +} + +impl BlockKey { + pub fn new(dva: Dva, objid: u64, block_index: u64) -> Self { + Self { + dva, + objid, + block_index, + } + } +} + +/// Cache statistics, for tests and metrics. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct CacheStats { + pub hits: u64, + pub misses: u64, + pub evictions: u64, +} + +#[derive(Debug)] +struct Inner { + used: u64, + /// Insertion order is age; entries move to the back on access. + lru: LinkedHashMap, + stats: CacheStats, +} + +/// In-memory cache of decrypted blocks. +#[derive(Debug)] +pub struct BlockCache { + limit: u64, + inner: Mutex, +} + +impl BlockCache { + pub fn new(limit_bytes: u64) -> Self { + Self { + limit: limit_bytes, + inner: Mutex::new(Inner { + used: 0, + lru: LinkedHashMap::new(), + stats: CacheStats::default(), + }), + } + } + + pub fn get(&self, key: &BlockKey) -> Option { + let mut g = self.inner.lock(); + match g.lru.raw_entry_mut().from_key(key) { + hashlink::linked_hash_map::RawEntryMut::Occupied(mut e) => { + e.to_back(); + let v = e.get().clone(); + g.stats.hits += 1; + Some(v) + } + hashlink::linked_hash_map::RawEntryMut::Vacant(_) => { + g.stats.misses += 1; + None + } + } + } + + /// Insert, evicting oldest-first until the entry fits. + /// + /// A block larger than the whole budget is simply not cached, rather than + /// evicting everything to make room for something that will immediately be + /// evicted itself. + pub fn insert(&self, key: BlockKey, block: Bytes) { + let len = block.len() as u64; + if len > self.limit { + return; + } + let mut g = self.inner.lock(); + if let Some(old) = g.lru.insert(key, block) { + g.used -= old.len() as u64; + } + g.used += len; + while g.used > self.limit { + match g.lru.pop_front() { + Some((_, evicted)) => { + g.used -= evicted.len() as u64; + g.stats.evictions += 1; + } + None => break, + } + } + } + + pub fn used_bytes(&self) -> u64 { + self.inner.lock().used + } + + pub fn len(&self) -> usize { + self.inner.lock().lru.len() + } + + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + + pub fn stats(&self) -> CacheStats { + self.inner.lock().stats + } + + #[cfg(test)] + fn contains(&self, key: &BlockKey) -> bool { + self.inner.lock().lru.contains_key(key) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn dva(offset: u32, len: u32) -> Dva { + Dva { + txg: 1, + slab: 0, + offset, + len, + } + } + + /// A key at the canonical position, for tests that only care about LRU + /// behaviour rather than about position binding. + fn k(offset: u32, len: u32) -> BlockKey { + BlockKey::new(dva(offset, len), 1, 0) + } + + fn block(n: usize) -> Bytes { + Bytes::from(vec![0u8; n]) + } + + #[test] + fn hit_and_miss() { + let c = BlockCache::new(1024); + assert!(c.get(&k(0, 10)).is_none()); + c.insert(k(0, 10), block(10)); + assert_eq!(c.get(&k(0, 10)).unwrap().len(), 10); + assert_eq!( + c.stats(), + CacheStats { + hits: 1, + misses: 1, + evictions: 0 + } + ); + } + + #[test] + fn accounts_bytes() { + let c = BlockCache::new(1024); + c.insert(k(0, 100), block(100)); + c.insert(k(100, 200), block(200)); + assert_eq!(c.used_bytes(), 300); + assert_eq!(c.len(), 2); + } + + #[test] + fn evicts_oldest_first() { + let c = BlockCache::new(300); + c.insert(k(0, 100), block(100)); + c.insert(k(1, 100), block(100)); + c.insert(k(2, 100), block(100)); + assert_eq!(c.used_bytes(), 300); + + c.insert(k(3, 100), block(100)); + assert_eq!(c.used_bytes(), 300); + assert!(!c.contains(&k(0, 100)), "oldest should have been evicted"); + assert!(c.contains(&k(3, 100))); + assert_eq!(c.stats().evictions, 1); + } + + #[test] + fn access_refreshes_recency() { + let c = BlockCache::new(300); + c.insert(k(0, 100), block(100)); + c.insert(k(1, 100), block(100)); + c.insert(k(2, 100), block(100)); + + // Touch the oldest so it is no longer the eviction candidate. + assert!(c.get(&k(0, 100)).is_some()); + c.insert(k(3, 100), block(100)); + + assert!(c.contains(&k(0, 100)), "touched entry must survive"); + assert!(!c.contains(&k(1, 100))); + } + + #[test] + fn oversized_block_is_not_cached() { + let c = BlockCache::new(100); + c.insert(k(0, 200), block(200)); + assert!(c.is_empty()); + assert_eq!(c.used_bytes(), 0); + } + + #[test] + fn reinsert_does_not_double_count() { + let c = BlockCache::new(1024); + c.insert(k(0, 100), block(100)); + c.insert(k(0, 100), block(100)); + assert_eq!(c.used_bytes(), 100); + assert_eq!(c.len(), 1); + } + + /// Copy-on-write is what makes this cache safe without invalidation: + /// two different versions of a block occupy two different addresses, so a + /// stale entry is unreachable rather than wrong. + #[test] + fn rewritten_blocks_get_distinct_keys() { + let c = BlockCache::new(1024); + let at = |txg| { + BlockKey::new( + Dva { + txg, + slab: 0, + offset: 0, + len: 100, + }, + 1, + 0, + ) + }; + c.insert(at(1), Bytes::from_static(b"old")); + c.insert(at(2), Bytes::from_static(b"new")); + assert_eq!(c.get(&at(1)).unwrap(), Bytes::from_static(b"old")); + assert_eq!(c.get(&at(2)).unwrap(), Bytes::from_static(b"new")); + } + + /// Regression guard. Keyed on the address alone, a block cached after a + /// legitimate read would be served straight back when a forged pointer + /// claimed it belonged somewhere else — silently skipping the AEAD's + /// position check. The position must be part of the key. + #[test] + fn the_same_address_at_a_different_position_is_a_miss() { + let c = BlockCache::new(1024); + let d = dva(0, 100); + c.insert(BlockKey::new(d, 10, 5), Bytes::from_static(b"payload")); + + assert!(c.get(&BlockKey::new(d, 10, 5)).is_some()); + assert!(c.get(&BlockKey::new(d, 11, 5)).is_none(), "wrong objid"); + assert!(c.get(&BlockKey::new(d, 10, 6)).is_none(), "wrong index"); + } + + #[test] + fn zero_budget_caches_nothing() { + let c = BlockCache::new(0); + c.insert(k(0, 10), block(10)); + assert!(c.is_empty()); + } +} diff --git a/crates/s3fs-core/src/store/config.rs b/crates/s3fs-core/src/store/config.rs new file mode 100644 index 0000000..81d6570 --- /dev/null +++ b/crates/s3fs-core/src/store/config.rs @@ -0,0 +1,201 @@ +//! Store tuning knobs and key-space layout. +//! +//! Kept separate from [`crate::config::Config`] while the old path→key engine +//! still exists; the two merge when that engine is removed. + +use std::time::Duration; + +use crate::errors::{FsError, FsResult}; + +/// Default record size. Matches ZFS's default for the same reasons: large +/// enough that per-block overhead (128-byte pointer, 16-byte AEAD tag, one +/// BLAKE3 pass) is negligible, small enough that rewriting one record after a +/// small write does not amplify badly. +pub const DEFAULT_RECORD_SIZE: usize = 128 * 1024; + +/// Default cap on a single slab object. A commit writes `ceil(dirty / +/// SLAB_MAX)` objects, so this trades PUT count against how much has to be +/// re-sent if one PUT fails. +pub const DEFAULT_SLAB_MAX_BYTES: usize = 64 * 1024 * 1024; + +/// Smallest and largest permitted record sizes. +const MIN_RECORD_SIZE: usize = 4 * 1024; +const MAX_RECORD_SIZE: usize = 1024 * 1024; + +/// How long a committed root is retained when per-object Object Lock is used. +/// Ten years, per the deployment contract. +pub const DEFAULT_ROOT_RETENTION: Duration = Duration::from_secs(10 * 365 * 24 * 60 * 60); + +/// Store configuration. +#[derive(Debug, Clone)] +pub struct StoreConfig { + /// Key prefix inside both buckets. Empty for bucket root. + pub prefix: String, + /// Size of a level-0 data block and of an indirect block. Power of two. + pub record_size: usize, + /// Cap on one slab object. + pub slab_max_bytes: usize, + /// Byte budget for the decrypted-block cache. + pub block_cache_bytes: u64, + /// Max concurrent slab PUTs during a commit. + pub max_parallel_slab_puts: usize, + /// How many links of the root hash chain to verify at mount. + /// + /// 1 is enough to detect a spliced history at the tip, which is the live + /// attack; walking the whole chain is an audit operation, not a mount-time + /// one, because it costs one GET per root and roots are never deleted. + pub root_chain_verify_depth: u32, + /// Retention to stamp on root records, or `None` to rely on the bucket's + /// default retention configuration. + pub root_retention: Option, +} + +impl Default for StoreConfig { + fn default() -> Self { + Self { + prefix: String::new(), + record_size: DEFAULT_RECORD_SIZE, + slab_max_bytes: DEFAULT_SLAB_MAX_BYTES, + block_cache_bytes: 64 * 1024 * 1024, + max_parallel_slab_puts: 8, + root_chain_verify_depth: 1, + root_retention: Some(DEFAULT_ROOT_RETENTION), + } + } +} + +impl StoreConfig { + pub fn validate(&self) -> FsResult<()> { + if !self.record_size.is_power_of_two() { + return Err(FsError::Invalid("record_size must be a power of two")); + } + if self.record_size < MIN_RECORD_SIZE || self.record_size > MAX_RECORD_SIZE { + return Err(FsError::Invalid("record_size out of range")); + } + if self.slab_max_bytes < self.record_size { + return Err(FsError::Invalid("slab_max_bytes below record_size")); + } + // A slab offset is a u32 in the block pointer. + if self.slab_max_bytes > u32::MAX as usize { + return Err(FsError::Invalid("slab_max_bytes exceeds u32 addressing")); + } + if self.max_parallel_slab_puts == 0 { + return Err(FsError::Invalid("max_parallel_slab_puts must be nonzero")); + } + Ok(()) + } + + /// `log2(record_size)`, as stored in a dnode. + pub fn record_shift(&self) -> u8 { + self.record_size.trailing_zeros() as u8 + } + + /// Key of a slab object. Zero-padded hex so S3's lexicographic ordering + /// matches numeric ordering, which is what makes a range scan over txgs + /// meaningful. + pub fn slab_key(&self, txg: u64, slab: u16) -> String { + format!("{}slabs/{:016x}/{:04x}", self.prefix, txg, slab) + } + + /// Key of a root record. + pub fn root_key(&self, seq: u64) -> String { + format!("{}roots/{:016x}", self.prefix, seq) + } + + /// Key of the (untrusted) tip hint. + pub fn root_hint_key(&self) -> String { + format!("{}roots/latest", self.prefix) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn defaults_validate() { + StoreConfig::default().validate().unwrap(); + } + + #[test] + fn record_shift_matches_record_size() { + let mut c = StoreConfig::default(); + assert_eq!(c.record_shift(), 17); // 128 KiB + c.record_size = 4096; + assert_eq!(c.record_shift(), 12); + assert_eq!(1usize << c.record_shift(), c.record_size); + } + + #[test] + fn rejects_non_power_of_two_record_size() { + let c = StoreConfig { + record_size: 100_000, + ..Default::default() + }; + assert!(c.validate().is_err()); + } + + #[test] + fn rejects_out_of_range_record_size() { + for size in [1024, 2 * 1024 * 1024] { + let c = StoreConfig { + record_size: size, + ..Default::default() + }; + assert!(c.validate().is_err(), "accepted record_size {size}"); + } + } + + #[test] + fn rejects_slab_smaller_than_a_record() { + let c = StoreConfig { + slab_max_bytes: 4096, + ..Default::default() + }; + assert!(c.validate().is_err()); + } + + /// A slab offset is a `u32` in the block pointer, so a slab larger than + /// 4 GiB would silently truncate addresses. + #[test] + fn rejects_slab_beyond_u32_addressing() { + let c = StoreConfig { + slab_max_bytes: u32::MAX as usize + 1, + ..Default::default() + }; + assert!(c.validate().is_err()); + } + + #[test] + fn keys_are_zero_padded_and_sort_numerically() { + let c = StoreConfig::default(); + assert_eq!(c.root_key(1), "roots/0000000000000001"); + assert_eq!(c.root_key(0x2a), "roots/000000000000002a"); + assert_eq!(c.slab_key(9, 3), "slabs/0000000000000009/0003"); + + // The property that matters: lexicographic order == numeric order, + // so a LIST or a range scan sees roots in sequence order. + let mut keys: Vec<_> = [300u64, 2, 41, 1].iter().map(|&s| c.root_key(s)).collect(); + keys.sort(); + assert_eq!( + keys, + vec![ + c.root_key(1), + c.root_key(2), + c.root_key(41), + c.root_key(300) + ] + ); + } + + #[test] + fn prefix_is_applied_to_every_key() { + let c = StoreConfig { + prefix: "tenant-a/".into(), + ..Default::default() + }; + assert!(c.root_key(1).starts_with("tenant-a/roots/")); + assert!(c.slab_key(1, 0).starts_with("tenant-a/slabs/")); + assert!(c.root_hint_key().starts_with("tenant-a/roots/")); + } +} diff --git a/crates/s3fs-core/src/store/dir.rs b/crates/s3fs-core/src/store/dir.rs new file mode 100644 index 0000000..319fb63 --- /dev/null +++ b/crates/s3fs-core/src/store/dir.rs @@ -0,0 +1,978 @@ +//! Directories — a two-level B+tree of name-ordered entries. +//! +//! A directory object's data blocks are: +//! +//! ```text +//! block 0 index: [(separator, leaf block), …] sorted by separator +//! block 1..N leaves: sorted runs of (name, objid, kind) +//! ``` +//! +//! Lookup binary-searches the index for the leaf whose separator range covers +//! the name, then binary-searches that leaf: two block reads, both of which +//! stay cached for a hot directory. Inserting dirties one leaf plus the index, +//! and a leaf that overflows splits in half — so an insert costs three blocks +//! and their indirect path, not a rewrite of the directory. +//! +//! Because leaves are name-ordered and the index is separator-ordered, +//! `read_dir` is a concatenation with no sort step. +//! +//! ## Why not extendible hashing +//! +//! The plan called for an extendible hash table, whose cheap-doubling property +//! depends on many slots pointing at one shared bucket block. That is +//! unavailable here: [`crate::crypto::BlockAad`] binds every block to its +//! block index, so one physical block genuinely cannot be read from two +//! positions — which is exactly the property that stops an attacker relocating +//! blocks, and not one worth weakening for a directory layout. Sharing would +//! have to be expressed indirectly anyway, through a level of mapping; a +//! separator index is that mapping, and it also yields sorted iteration for +//! free. + +use std::collections::{BTreeMap, BTreeSet}; + +use crate::errors::{FsError, FsResult}; + +use super::blockstore::BlockStore; +use super::dnode::{Dnode, DnodeKind}; +use super::indirect::{commit_object, read_raw_block}; +use super::slab::SlabWriter; + +const INDEX_MAGIC: u32 = 0x4449_5831; // "DIX1" +const LEAF_MAGIC: u32 = 0x444C_4631; // "DLF1" + +/// The index always lives at data block 0; leaves start at 1. +const INDEX_BLOCK: u64 = 0; + +/// Longest permitted entry name, matching the path validator. +pub const MAX_NAME_LEN: usize = 255; + +/// One directory entry. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Dirent { + pub name: String, + pub objid: u64, + pub kind: DnodeKind, +} + +impl Dirent { + fn encoded_len(&self) -> usize { + 12 + self.name.len() + } + + fn encode(&self, out: &mut Vec) { + out.extend_from_slice(&(self.name.len() as u16).to_be_bytes()); + out.push(self.kind as u8); + out.push(0); // reserved + out.extend_from_slice(&self.objid.to_be_bytes()); + out.extend_from_slice(self.name.as_bytes()); + } +} + +/// Cursor over a byte slice that refuses to read past the end. +struct Reader<'a> { + buf: &'a [u8], + pos: usize, +} + +impl<'a> Reader<'a> { + fn new(buf: &'a [u8]) -> Self { + Reader { buf, pos: 0 } + } + + fn take(&mut self, n: usize, what: &'static str) -> FsResult<&'a [u8]> { + let end = self.pos.checked_add(n).ok_or(FsError::Integrity(what))?; + let slice = self + .buf + .get(self.pos..end) + .ok_or(FsError::Integrity(what))?; + self.pos = end; + Ok(slice) + } + + fn u16(&mut self, what: &'static str) -> FsResult { + Ok(u16::from_be_bytes( + self.take(2, what)?.try_into().expect("2 bytes"), + )) + } + + fn u32(&mut self, what: &'static str) -> FsResult { + Ok(u32::from_be_bytes( + self.take(4, what)?.try_into().expect("4 bytes"), + )) + } + + fn u64(&mut self, what: &'static str) -> FsResult { + Ok(u64::from_be_bytes( + self.take(8, what)?.try_into().expect("8 bytes"), + )) + } +} + +fn encode_leaf(entries: &[Dirent]) -> Vec { + let cap = 8 + entries.iter().map(Dirent::encoded_len).sum::(); + let mut out = Vec::with_capacity(cap); + out.extend_from_slice(&LEAF_MAGIC.to_be_bytes()); + out.extend_from_slice(&(entries.len() as u32).to_be_bytes()); + for e in entries { + e.encode(&mut out); + } + out +} + +fn leaf_encoded_len(entries: &[Dirent]) -> usize { + 8 + entries.iter().map(Dirent::encoded_len).sum::() +} + +fn decode_leaf(buf: &[u8]) -> FsResult> { + const BAD: &str = "dir: malformed leaf block"; + let mut r = Reader::new(buf); + if r.u32(BAD)? != LEAF_MAGIC { + return Err(FsError::Integrity("dir: bad leaf magic")); + } + let count = r.u32(BAD)? as usize; + let mut out: Vec = Vec::with_capacity(count.min(4096)); + for _ in 0..count { + let name_len = r.u16(BAD)? as usize; + let kind = r.take(1, BAD)?[0]; + if r.take(1, BAD)?[0] != 0 { + return Err(FsError::Integrity("dir: entry reserved byte not zero")); + } + let objid = r.u64(BAD)?; + let name = std::str::from_utf8(r.take(name_len, BAD)?) + .map_err(|_| FsError::Integrity("dir: entry name is not valid UTF-8"))? + .to_string(); + out.push(Dirent { + name, + objid, + kind: decode_kind(kind)?, + }); + } + if r.pos != buf.len() { + return Err(FsError::Integrity("dir: trailing bytes in leaf block")); + } + // Strictly increasing order is what makes binary search sound, and it also + // rules out duplicate names — which a forged block could otherwise use to + // make one name resolve to two different objects. + if out.windows(2).any(|w| w[0].name >= w[1].name) { + return Err(FsError::Integrity("dir: leaf entries are not sorted")); + } + Ok(out) +} + +fn decode_kind(v: u8) -> FsResult { + Ok(match v { + 1 => DnodeKind::File, + 2 => DnodeKind::Dir, + 3 => DnodeKind::Symlink, + _ => return Err(FsError::Integrity("dir: entry has a non-linkable kind")), + }) +} + +/// A leaf and the smallest name that may live in it. +/// +/// The separator is a boundary, not necessarily a name that exists: entries +/// are deleted without disturbing it, so an empty leaf keeps its place in the +/// ordering. +#[derive(Debug, Clone, PartialEq, Eq)] +struct LeafRef { + sep: String, + block: u64, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +struct DirIndex { + entry_count: u64, + leaves: Vec, +} + +impl DirIndex { + /// A directory with one empty leaf whose separator matches everything. + fn empty() -> Self { + DirIndex { + entry_count: 0, + leaves: vec![LeafRef { + sep: String::new(), + block: 1, + }], + } + } + + fn encode(&self) -> Vec { + let mut out = Vec::with_capacity(16 + self.leaves.len() * 24); + out.extend_from_slice(&INDEX_MAGIC.to_be_bytes()); + out.extend_from_slice(&(self.leaves.len() as u32).to_be_bytes()); + out.extend_from_slice(&self.entry_count.to_be_bytes()); + for l in &self.leaves { + out.extend_from_slice(&l.block.to_be_bytes()); + out.extend_from_slice(&(l.sep.len() as u16).to_be_bytes()); + out.extend_from_slice(l.sep.as_bytes()); + } + out + } + + fn decode(buf: &[u8]) -> FsResult { + const BAD: &str = "dir: malformed index block"; + let mut r = Reader::new(buf); + if r.u32(BAD)? != INDEX_MAGIC { + return Err(FsError::Integrity("dir: bad index magic")); + } + let leaf_count = r.u32(BAD)? as usize; + let entry_count = r.u64(BAD)?; + let mut leaves = Vec::with_capacity(leaf_count.min(4096)); + for _ in 0..leaf_count { + let block = r.u64(BAD)?; + let sep_len = r.u16(BAD)? as usize; + let sep = std::str::from_utf8(r.take(sep_len, BAD)?) + .map_err(|_| FsError::Integrity("dir: separator is not valid UTF-8"))? + .to_string(); + leaves.push(LeafRef { sep, block }); + } + if r.pos != buf.len() { + return Err(FsError::Integrity("dir: trailing bytes in index block")); + } + if leaves.is_empty() { + return Err(FsError::Integrity("dir: index has no leaves")); + } + if !leaves[0].sep.is_empty() { + return Err(FsError::Integrity("dir: first separator is not empty")); + } + if leaves.windows(2).any(|w| w[0].sep >= w[1].sep) { + return Err(FsError::Integrity("dir: separators are not sorted")); + } + if leaves.iter().any(|l| l.block == INDEX_BLOCK) { + return Err(FsError::Integrity( + "dir: leaf collides with the index block", + )); + } + let mut seen = BTreeSet::new(); + if !leaves.iter().all(|l| seen.insert(l.block)) { + return Err(FsError::Integrity("dir: duplicate leaf block")); + } + Ok(DirIndex { + entry_count, + leaves, + }) + } + + /// Position of the leaf that owns `name`: the last one whose separator + /// does not exceed it. + fn position_of(&self, name: &str) -> usize { + match self.leaves.binary_search_by(|l| l.sep.as_str().cmp(name)) { + Ok(i) => i, + // `partition_point`-style: the insertion point is one past the + // owning leaf. Index 0 has an empty separator, so this never + // underflows. + Err(i) => i - 1, + } + } + + fn next_block(&self) -> u64 { + self.leaves.iter().map(|l| l.block).max().unwrap_or(0) + 1 + } +} + +/// A staged set of changes to one directory. +/// +/// Leaves are loaded on demand, so a lookup or a single insert touches one +/// leaf regardless of how large the directory is. Call [`DirTxn::finish`] to +/// stage the modified blocks into a transaction group. +#[derive(Debug)] +pub struct DirTxn<'a> { + bs: &'a BlockStore, + dnode: Dnode, + index: DirIndex, + loaded: BTreeMap>, + dirty: BTreeSet, +} + +impl<'a> DirTxn<'a> { + /// Open a directory for reading and modification. + pub async fn load(bs: &'a BlockStore, dnode: &Dnode) -> FsResult> { + if dnode.kind != DnodeKind::Dir { + return Err(FsError::NotDirectory); + } + let (index, dirty) = match read_raw_block(bs, dnode, INDEX_BLOCK).await? { + Some(raw) => (DirIndex::decode(&raw)?, BTreeSet::new()), + // A directory that has never been written has no blocks at all. + // Synthesise the empty shape and mark it dirty so the first commit + // materialises it. + None => (DirIndex::empty(), BTreeSet::from([INDEX_BLOCK, 1])), + }; + let mut txn = DirTxn { + bs, + dnode: dnode.clone(), + index, + loaded: BTreeMap::new(), + dirty, + }; + if txn.dirty.contains(&1) { + txn.loaded.insert(1, Vec::new()); + } + Ok(txn) + } + + pub fn entry_count(&self) -> u64 { + self.index.entry_count + } + + async fn leaf(&mut self, block: u64) -> FsResult<&mut Vec> { + if !self.loaded.contains_key(&block) { + let entries = match read_raw_block(self.bs, &self.dnode, block).await? { + Some(raw) => decode_leaf(&raw)?, + None => Vec::new(), + }; + self.loaded.insert(block, entries); + } + Ok(self.loaded.get_mut(&block).expect("just inserted")) + } + + /// Find one entry by name. + pub async fn lookup(&mut self, name: &str) -> FsResult> { + let block = self.index.leaves[self.index.position_of(name)].block; + let entries = self.leaf(block).await?; + Ok(entries + .binary_search_by(|e| e.name.as_str().cmp(name)) + .ok() + .map(|i| entries[i].clone())) + } + + /// Add an entry. Fails with [`FsError::AlreadyExists`] if the name is taken. + pub async fn insert(&mut self, entry: Dirent) -> FsResult<()> { + if entry.name.is_empty() || entry.name.len() > MAX_NAME_LEN { + return Err(FsError::NameTooLong); + } + if entry.kind == DnodeKind::Free || entry.kind == DnodeKind::DnodeArray { + return Err(FsError::Invalid("directory entry has a non-linkable kind")); + } + + let record_size = self.dnode.record_size(); + let pos = self.index.position_of(&entry.name); + let block = self.index.leaves[pos].block; + let entries = self.leaf(block).await?; + + let at = match entries.binary_search_by(|e| e.name.cmp(&entry.name)) { + Ok(_) => return Err(FsError::AlreadyExists), + Err(at) => at, + }; + entries.insert(at, entry); + let overflowed = leaf_encoded_len(entries) > record_size; + + self.dirty.insert(block); + self.index.entry_count += 1; + if overflowed { + self.split_leaf(pos, record_size)?; + } + Ok(()) + } + + /// Split the leaf at index `pos` in half. + /// + /// The new leaf takes the upper half and its separator is that half's first + /// name, so every existing name still resolves to the leaf that holds it. + fn split_leaf(&mut self, pos: usize, record_size: usize) -> FsResult<()> { + let block = self.index.leaves[pos].block; + let entries = self.loaded.get_mut(&block).expect("leaf is loaded"); + if entries.len() < 2 { + // A single entry that does not fit cannot be split apart. With + // names capped at 255 bytes and records at 4 KiB or more, this is + // unreachable; refusing beats looping. + return Err(FsError::FileTooLarge); + } + let mid = entries.len() / 2; + let upper: Vec = entries.split_off(mid); + let sep = upper[0].name.clone(); + let new_block = self.index.next_block(); + + self.loaded.insert(new_block, upper); + self.dirty.insert(new_block); + self.index.leaves.insert( + pos + 1, + LeafRef { + sep, + block: new_block, + }, + ); + + if self.index.encode().len() > record_size { + return Err(FsError::FileTooLarge); + } + Ok(()) + } + + /// Remove an entry by name, returning it. + /// + /// An emptied leaf keeps its slot rather than being merged away. Its + /// separator still partitions the name space correctly, it costs eight + /// stored bytes, and a later insert in that range refills it. + pub async fn remove(&mut self, name: &str) -> FsResult { + let block = self.index.leaves[self.index.position_of(name)].block; + let entries = self.leaf(block).await?; + let at = entries + .binary_search_by(|e| e.name.as_str().cmp(name)) + .map_err(|_| FsError::NotFound)?; + let removed = entries.remove(at); + self.dirty.insert(block); + self.index.entry_count -= 1; + Ok(removed) + } + + /// Every entry, in name order. + pub async fn list(&mut self) -> FsResult> { + let blocks: Vec = self.index.leaves.iter().map(|l| l.block).collect(); + let mut out = Vec::with_capacity(self.index.entry_count as usize); + for block in blocks { + out.extend_from_slice(self.leaf(block).await?); + } + Ok(out) + } + + /// `true` if the directory holds no entries. Used by `rmdir`. + pub fn is_empty(&self) -> bool { + self.index.entry_count == 0 + } + + /// Stage every modified block into `writer` and return the updated dnode. + pub async fn finish(mut self, writer: &mut SlabWriter) -> FsResult { + if self.dirty.is_empty() { + return Ok(self.dnode); + } + self.dirty.insert(INDEX_BLOCK); + + let mut blocks: BTreeMap> = BTreeMap::new(); + for block in &self.dirty { + let bytes = if *block == INDEX_BLOCK { + self.index.encode() + } else { + encode_leaf(self.loaded.get(block).map(Vec::as_slice).unwrap_or(&[])) + }; + blocks.insert(*block, bytes); + } + + // Blocks 0..=highest leaf; the geometry only needs to cover them. + let nblocks = self.index.next_block(); + let new_size = nblocks * self.dnode.record_size() as u64; + commit_object(self.bs, writer, &self.dnode, blocks, new_size).await + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::backend::memory::MemoryBackend; + use crate::crypto::{KeyMaterial, MasterSecret}; + use crate::store::config::StoreConfig; + use std::sync::Arc; + + fn store(record_size: usize) -> BlockStore { + let backend = Arc::new(MemoryBackend::new()); + let keys = + Arc::new(KeyMaterial::derive(&MasterSecret::from_bytes([8u8; 32]), [0u8; 16]).unwrap()); + BlockStore::new( + backend, + keys, + Arc::new(StoreConfig { + record_size, + ..Default::default() + }), + ) + } + + fn new_dir(record_shift: u8) -> Dnode { + Dnode::new(7, DnodeKind::Dir, record_shift, 0) + } + + fn entry(name: &str, objid: u64) -> Dirent { + Dirent { + name: name.to_string(), + objid, + kind: DnodeKind::File, + } + } + + /// Apply a closure to the directory and commit it as one transaction group. + async fn apply(bs: &BlockStore, txg: u64, dnode: &Dnode, f: F) -> Dnode + where + F: for<'x> FnOnce( + &'x mut DirTxn<'_>, + ) -> std::pin::Pin< + Box> + 'x>, + >, + { + let mut txn = DirTxn::load(bs, dnode).await.unwrap(); + f(&mut txn).await.unwrap(); + let mut w = SlabWriter::new(txg, bs.config()); + let out = txn.finish(&mut w).await.unwrap(); + bs.write_slabs(txg, w.finish().unwrap()).await.unwrap(); + out + } + + async fn insert_all(bs: &BlockStore, txg: u64, dnode: &Dnode, names: &[&str]) -> Dnode { + let mut txn = DirTxn::load(bs, dnode).await.unwrap(); + for (i, n) in names.iter().enumerate() { + txn.insert(entry(n, 100 + i as u64)).await.unwrap(); + } + let mut w = SlabWriter::new(txg, bs.config()); + let out = txn.finish(&mut w).await.unwrap(); + bs.write_slabs(txg, w.finish().unwrap()).await.unwrap(); + out + } + + async fn names_of(bs: &BlockStore, dnode: &Dnode) -> Vec { + let mut txn = DirTxn::load(bs, dnode).await.unwrap(); + txn.list() + .await + .unwrap() + .into_iter() + .map(|e| e.name) + .collect() + } + + #[tokio::test] + async fn a_fresh_directory_is_empty() { + let bs = store(4096); + let d = new_dir(12); + let mut txn = DirTxn::load(&bs, &d).await.unwrap(); + assert!(txn.is_empty()); + assert_eq!(txn.list().await.unwrap(), vec![]); + assert_eq!(txn.lookup("anything").await.unwrap(), None); + } + + #[tokio::test] + async fn insert_then_look_up() { + let bs = store(4096); + let d = insert_all(&bs, 1, &new_dir(12), &["hello.txt"]).await; + + let mut txn = DirTxn::load(&bs, &d).await.unwrap(); + assert_eq!( + txn.lookup("hello.txt").await.unwrap(), + Some(entry("hello.txt", 100)) + ); + assert_eq!(txn.lookup("missing").await.unwrap(), None); + assert_eq!(txn.entry_count(), 1); + } + + #[tokio::test] + async fn entries_come_back_in_name_order() { + let bs = store(4096); + let d = insert_all(&bs, 1, &new_dir(12), &["zebra", "apple", "Mango", "banana"]).await; + assert_eq!( + names_of(&bs, &d).await, + vec!["Mango", "apple", "banana", "zebra"], + "byte order, so uppercase sorts first" + ); + } + + #[tokio::test] + async fn duplicate_names_are_rejected() { + let bs = store(4096); + let d = insert_all(&bs, 1, &new_dir(12), &["dup"]).await; + + let mut txn = DirTxn::load(&bs, &d).await.unwrap(); + assert!(matches!( + txn.insert(entry("dup", 999)).await, + Err(FsError::AlreadyExists) + )); + } + + #[tokio::test] + async fn remove_takes_the_entry_out() { + let bs = store(4096); + let d = insert_all(&bs, 1, &new_dir(12), &["a", "b", "c"]).await; + let d = apply(&bs, 2, &d, |t| { + Box::pin(async move { + t.remove("b").await?; + Ok(()) + }) + }) + .await; + + assert_eq!(names_of(&bs, &d).await, vec!["a", "c"]); + let mut txn = DirTxn::load(&bs, &d).await.unwrap(); + assert_eq!(txn.lookup("b").await.unwrap(), None); + assert_eq!(txn.entry_count(), 2); + assert!(matches!(txn.remove("b").await, Err(FsError::NotFound))); + } + + #[tokio::test] + async fn removing_everything_leaves_an_empty_directory() { + let bs = store(4096); + let d = insert_all(&bs, 1, &new_dir(12), &["only"]).await; + let d = apply(&bs, 2, &d, |t| { + Box::pin(async move { + t.remove("only").await?; + Ok(()) + }) + }) + .await; + + let txn = DirTxn::load(&bs, &d).await.unwrap(); + assert!(txn.is_empty(), "rmdir depends on this"); + assert_eq!(names_of(&bs, &d).await, Vec::::new()); + } + + #[tokio::test] + async fn names_are_validated() { + let bs = store(4096); + let d = new_dir(12); + let mut txn = DirTxn::load(&bs, &d).await.unwrap(); + + assert!(matches!( + txn.insert(entry("", 1)).await, + Err(FsError::NameTooLong) + )); + let long = "x".repeat(MAX_NAME_LEN + 1); + assert!(matches!( + txn.insert(entry(&long, 1)).await, + Err(FsError::NameTooLong) + )); + assert!(txn + .insert(entry(&"x".repeat(MAX_NAME_LEN), 1)) + .await + .is_ok()); + } + + #[tokio::test] + async fn a_directory_entry_cannot_point_at_a_free_slot() { + let bs = store(4096); + let d = new_dir(12); + let mut txn = DirTxn::load(&bs, &d).await.unwrap(); + for kind in [DnodeKind::Free, DnodeKind::DnodeArray] { + assert!(txn + .insert(Dirent { + name: "x".into(), + objid: 5, + kind + }) + .await + .is_err()); + } + } + + // ---- splitting --------------------------------------------------------- + + #[tokio::test] + async fn a_leaf_that_overflows_splits() { + let bs = store(4096); + // ~40 bytes per entry, so a 4 KiB leaf holds roughly 100. + let names: Vec = (0..300).map(|i| format!("file-{i:04}-padding")).collect(); + let refs: Vec<&str> = names.iter().map(String::as_str).collect(); + let d = insert_all(&bs, 1, &new_dir(12), &refs).await; + + let mut expected = names.clone(); + expected.sort(); + assert_eq!(names_of(&bs, &d).await, expected); + + // Every name must still be individually reachable through the index. + let mut txn = DirTxn::load(&bs, &d).await.unwrap(); + assert!(txn.index.leaves.len() > 1, "the leaf should have split"); + for n in &names { + assert!(txn.lookup(n).await.unwrap().is_some(), "lost {n}"); + } + assert_eq!(txn.entry_count(), 300); + } + + #[tokio::test] + async fn splits_survive_across_transaction_groups() { + let bs = store(4096); + let mut d = new_dir(12); + let mut all = Vec::new(); + for txg in 1..=6u64 { + let names: Vec = (0..50) + .map(|i| format!("g{txg}-entry-{i:03}-padding")) + .collect(); + let refs: Vec<&str> = names.iter().map(String::as_str).collect(); + d = insert_all(&bs, txg, &d, &refs).await; + all.extend(names); + } + + all.sort(); + assert_eq!(names_of(&bs, &d).await, all); + + let mut txn = DirTxn::load(&bs, &d).await.unwrap(); + assert_eq!(txn.entry_count(), 300); + for n in &all { + assert!(txn.lookup(n).await.unwrap().is_some(), "lost {n}"); + } + } + + #[tokio::test] + async fn an_emptied_leaf_still_accepts_inserts() { + let bs = store(4096); + let names: Vec = (0..300).map(|i| format!("file-{i:04}-padding")).collect(); + let refs: Vec<&str> = names.iter().map(String::as_str).collect(); + let d = insert_all(&bs, 1, &new_dir(12), &refs).await; + + // Empty one leaf's worth of the name space, then refill part of it. + let d = apply(&bs, 2, &d, |t| { + Box::pin(async move { + for i in 0..300 { + t.remove(&format!("file-{i:04}-padding")).await?; + } + Ok(()) + }) + }) + .await; + assert!(DirTxn::load(&bs, &d).await.unwrap().is_empty()); + + let d = insert_all(&bs, 3, &d, &["file-0100-padding", "file-0200-padding"]).await; + assert_eq!( + names_of(&bs, &d).await, + vec!["file-0100-padding", "file-0200-padding"] + ); + } + + // ---- copy-on-write ----------------------------------------------------- + + #[tokio::test] + async fn a_superseded_directory_still_reads() { + let bs = store(4096); + let old = insert_all(&bs, 1, &new_dir(12), &["a", "b"]).await; + let new = insert_all(&bs, 2, &old, &["c"]).await; + + assert_eq!(names_of(&bs, &new).await, vec!["a", "b", "c"]); + assert_eq!( + names_of(&bs, &old).await, + vec!["a", "b"], + "the previous version of the directory must be unchanged" + ); + } + + #[tokio::test] + async fn a_no_op_transaction_does_not_rewrite_the_directory() { + let bs = store(4096); + let d = insert_all(&bs, 1, &new_dir(12), &["a"]).await; + + let txn = DirTxn::load(&bs, &d).await.unwrap(); + let mut w = SlabWriter::new(2, bs.config()); + let same = txn.finish(&mut w).await.unwrap(); + assert_eq!(same.blkptr, d.blkptr); + assert_eq!(w.block_count(), 0); + } + + #[tokio::test] + async fn loading_a_non_directory_is_refused() { + let bs = store(4096); + let f = Dnode::new(3, DnodeKind::File, 12, 0); + assert!(matches!( + DirTxn::load(&bs, &f).await, + Err(FsError::NotDirectory) + )); + } + + // ---- encoding and tamper resistance ------------------------------------ + + #[test] + fn leaf_round_trips() { + let entries = vec![ + entry("alpha", 1), + Dirent { + name: "beta".into(), + objid: 2, + kind: DnodeKind::Dir, + }, + Dirent { + name: "gamma".into(), + objid: 3, + kind: DnodeKind::Symlink, + }, + ]; + let encoded = encode_leaf(&entries); + assert_eq!(encoded.len(), leaf_encoded_len(&entries)); + assert_eq!(decode_leaf(&encoded).unwrap(), entries); + } + + #[test] + fn empty_leaf_round_trips() { + assert_eq!(decode_leaf(&encode_leaf(&[])).unwrap(), vec![]); + } + + #[test] + fn index_round_trips() { + let idx = DirIndex { + entry_count: 9, + leaves: vec![ + LeafRef { + sep: String::new(), + block: 1, + }, + LeafRef { + sep: "m".into(), + block: 2, + }, + ], + }; + assert_eq!(DirIndex::decode(&idx.encode()).unwrap(), idx); + } + + /// Unsorted entries would break binary search, and duplicates would let one + /// name resolve to two objects depending on which the search happened to + /// land on. + #[test] + fn unsorted_or_duplicated_leaf_entries_are_rejected() { + let unsorted = encode_leaf(&[entry("b", 1), entry("a", 2)]); + assert!(matches!( + decode_leaf(&unsorted), + Err(FsError::Integrity("dir: leaf entries are not sorted")) + )); + + let duplicated = encode_leaf(&[entry("a", 1), entry("a", 2)]); + assert!(decode_leaf(&duplicated).is_err()); + } + + #[test] + fn malformed_blocks_are_rejected() { + assert!(decode_leaf(&[]).is_err()); + assert!(decode_leaf(&[0u8; 4]).is_err()); + assert!(DirIndex::decode(&[]).is_err()); + + // Wrong magic. + let mut leaf = encode_leaf(&[entry("a", 1)]); + leaf[0] ^= 0xff; + assert!(matches!( + decode_leaf(&leaf), + Err(FsError::Integrity("dir: bad leaf magic")) + )); + + // A count that overruns the buffer. + let mut leaf = encode_leaf(&[entry("a", 1)]); + leaf[7] = 9; + assert!(decode_leaf(&leaf).is_err()); + + // Trailing bytes: everything in the block must be accounted for. + let mut leaf = encode_leaf(&[entry("a", 1)]); + leaf.push(0); + assert!(matches!( + decode_leaf(&leaf), + Err(FsError::Integrity("dir: trailing bytes in leaf block")) + )); + } + + #[test] + fn index_invariants_are_enforced() { + let mk = |leaves: Vec| DirIndex { + entry_count: 0, + leaves, + }; + + // No leaves at all. + assert!(DirIndex::decode(&mk(vec![]).encode()).is_err()); + // First separator must be empty or some names resolve nowhere. + assert!(DirIndex::decode( + &mk(vec![LeafRef { + sep: "m".into(), + block: 1 + }]) + .encode() + ) + .is_err()); + // Out-of-order separators break binary search. + assert!(DirIndex::decode( + &mk(vec![ + LeafRef { + sep: String::new(), + block: 1 + }, + LeafRef { + sep: "z".into(), + block: 2 + }, + LeafRef { + sep: "a".into(), + block: 3 + }, + ]) + .encode() + ) + .is_err()); + // A leaf pointing at block 0 would alias the index itself. + assert!(DirIndex::decode( + &mk(vec![LeafRef { + sep: String::new(), + block: 0 + }]) + .encode() + ) + .is_err()); + // Two separators pointing at one block. + assert!(DirIndex::decode( + &mk(vec![ + LeafRef { + sep: String::new(), + block: 1 + }, + LeafRef { + sep: "m".into(), + block: 1 + }, + ]) + .encode() + ) + .is_err()); + } + + #[test] + fn separator_search_finds_the_owning_leaf() { + let idx = DirIndex { + entry_count: 0, + leaves: vec![ + LeafRef { + sep: String::new(), + block: 1, + }, + LeafRef { + sep: "m".into(), + block: 2, + }, + LeafRef { + sep: "t".into(), + block: 3, + }, + ], + }; + assert_eq!(idx.position_of(""), 0); + assert_eq!(idx.position_of("a"), 0); + assert_eq!(idx.position_of("lzzz"), 0); + assert_eq!(idx.position_of("m"), 1, "exact separator match"); + assert_eq!(idx.position_of("s"), 1); + assert_eq!(idx.position_of("t"), 2); + assert_eq!(idx.position_of("zzz"), 2); + } + + #[tokio::test] + async fn a_tampered_directory_block_is_rejected() { + use crate::backend::{Backend, PutBlobInput}; + let backend = Arc::new(MemoryBackend::new()); + let keys = + Arc::new(KeyMaterial::derive(&MasterSecret::from_bytes([8u8; 32]), [0u8; 16]).unwrap()); + let bs = BlockStore::new( + backend.clone(), + keys, + Arc::new(StoreConfig { + record_size: 4096, + ..Default::default() + }), + ); + + let d = insert_all(&bs, 1, &new_dir(12), &["a", "b"]).await; + + let key = bs.config().slab_key(1, 0); + let mut body = backend.get_blob(&key, None).await.unwrap().body.to_vec(); + body[0] ^= 0xff; + backend + .put_blob(PutBlobInput::new(key, bytes::Bytes::from(body))) + .await + .unwrap(); + + // Whether the corruption is caught opening the index or reading a leaf + // is an implementation detail; that no entries come back is not. + let result = async { + let mut txn = DirTxn::load(&bs, &d).await?; + txn.list().await + } + .await; + assert!( + matches!(result, Err(FsError::Integrity(_))), + "expected an integrity failure, got {result:?}" + ); + } +} diff --git a/crates/s3fs-core/src/store/dnode.rs b/crates/s3fs-core/src/store/dnode.rs new file mode 100644 index 0000000..a263423 --- /dev/null +++ b/crates/s3fs-core/src/store/dnode.rs @@ -0,0 +1,638 @@ +//! `Dnode` — the on-disk inode. +//! +//! One dnode per file, directory, or symlink. It holds all POSIX metadata and +//! a single block pointer that roots the object's indirect tree. There is no +//! S3 user metadata anywhere in this design: `set_times` is a field write, not +//! a `CopyObject` with `MetadataDirective=REPLACE`. +//! +//! ## Wire layout — 512 bytes, big-endian +//! +//! ```text +//! offset size field +//! 0 8 objid +//! 8 1 kind Free | File | Dir | Symlink | DnodeArray +//! 9 1 nlevels height of the indirect tree, data level included +//! 10 1 record_shift log2(record size) +//! 11 1 reserved must be zero +//! 12 4 nlink +//! 16 4 mode +//! 20 4 uid +//! 24 4 gid +//! 28 8 size logical size in bytes +//! 36 8 atime_nanos +//! 44 8 mtime_nanos +//! 52 8 ctime_nanos +//! 60 8 btime_nanos +//! 68 8 gen bumped on reuse; with objid, a stable identity +//! 76 2 inline_len +//! 78 2 reserved must be zero +//! 80 8 parent_objid containing directory; self for the root +//! 88 128 blkptr root of the indirect tree +//! 216 294 inline short symlink targets +//! 510 2 reserved must be zero +//! ``` +//! +//! 512 bytes divides the 128 KiB record exactly 256 ways, so a dnode never +//! straddles a block boundary and object id arithmetic is a shift. +//! +//! ## Tree height +//! +//! `nlevels` counts the data level: +//! +//! - `0` — no data at all; the pointer is a hole. +//! - `1` — the pointer *is* the single data block. +//! - `n` — the pointer is a level-`n-1` indirect block. +//! +//! One embedded pointer rather than ZFS's three: the extra two only help +//! objects of exactly two or three blocks, and dropping them buys 256 bytes of +//! inline space, which is what lets almost every symlink target live in the +//! dnode itself. + +use crate::errors::{FsError, FsResult}; + +use super::blkptr::{BlkPtr, BLKPTR_LEN, MAX_LEVEL}; + +/// Encoded size of a dnode. +pub const DNODE_LEN: usize = 512; + +/// Bytes of in-dnode storage for short symlink targets. +pub const INLINE_CAP: usize = 294; + +const PARENT_OFFSET: usize = 80; +const BLKPTR_OFFSET: usize = PARENT_OFFSET + 8; +const INLINE_OFFSET: usize = BLKPTR_OFFSET + BLKPTR_LEN; +const RESERVED_KIND_PAD: usize = 11; +const RESERVED0: std::ops::Range = 78..80; +const RESERVED1: std::ops::Range = 510..512; + +/// Deepest tree we accept: the root pointer sits at `nlevels - 1`, which must +/// itself be a decodable level. +pub const MAX_NLEVELS: u8 = MAX_LEVEL + 1; + +/// Object id of the meta-dnode — the dnode array that holds every other +/// dnode. It is never stored *in* the array; its pointer lives in the root +/// record. +pub const META_OBJID: u64 = 0; + +/// Object id of the root directory. +pub const ROOT_OBJID: u64 = 1; + +/// How many dnodes fit in one block. +pub const fn dnodes_per_block(record_size: usize) -> usize { + record_size / DNODE_LEN +} + +/// What an object is. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +#[repr(u8)] +pub enum DnodeKind { + /// Unallocated slot in the dnode array. + #[default] + Free = 0, + File = 1, + Dir = 2, + Symlink = 3, + /// The dnode array itself. + DnodeArray = 4, +} + +impl DnodeKind { + fn from_u8(v: u8) -> FsResult { + Ok(match v { + 0 => DnodeKind::Free, + 1 => DnodeKind::File, + 2 => DnodeKind::Dir, + 3 => DnodeKind::Symlink, + 4 => DnodeKind::DnodeArray, + _ => return Err(FsError::Integrity("dnode: unknown kind")), + }) + } +} + +/// An object's metadata and the root of its block tree. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Dnode { + pub objid: u64, + pub kind: DnodeKind, + pub nlevels: u8, + pub record_shift: u8, + /// Containing directory. The root directory is its own parent, which is + /// what makes `..` terminate there instead of escaping the mount. + /// + /// Held in the dnode rather than as a `..` directory entry so that + /// `read_dir` needs no filtering, `rmdir`'s emptiness check stays honest, + /// and a hard link to a directory remains inexpressible. + pub parent_objid: u64, + pub nlink: u32, + pub mode: u32, + pub uid: u32, + pub gid: u32, + pub size: u64, + pub atime_nanos: u64, + pub mtime_nanos: u64, + pub ctime_nanos: u64, + pub btime_nanos: u64, + /// Incremented when an object id is reused. `(objid, gen)` is a stable + /// identity across mounts, which is what makes `is-same-object` and + /// `metadata-hash` exact rather than a guess derived from an ETag. + pub gen: u64, + /// Short symlink target, stored in the dnode to avoid a block read. + pub inline: Vec, + /// Root of the indirect tree. + pub blkptr: BlkPtr, +} + +impl Dnode { + /// A free slot. + pub fn free(objid: u64, record_shift: u8) -> Self { + Dnode { + objid, + kind: DnodeKind::Free, + nlevels: 0, + record_shift, + parent_objid: objid, + nlink: 0, + mode: 0, + uid: 0, + gid: 0, + size: 0, + atime_nanos: 0, + mtime_nanos: 0, + ctime_nanos: 0, + btime_nanos: 0, + gen: 0, + inline: Vec::new(), + blkptr: BlkPtr::HOLE, + } + } + + /// A newly allocated object of the given kind. + pub fn new(objid: u64, kind: DnodeKind, record_shift: u8, now_nanos: u64) -> Self { + Dnode { + kind, + nlink: u32::from(kind != DnodeKind::Free), + atime_nanos: now_nanos, + mtime_nanos: now_nanos, + ctime_nanos: now_nanos, + btime_nanos: now_nanos, + ..Dnode::free(objid, record_shift) + } + } + + pub fn record_size(&self) -> usize { + 1usize << self.record_shift + } + + /// Pointers per indirect block at this object's record size. + pub fn fanout(&self) -> u64 { + (self.record_size() / BLKPTR_LEN) as u64 + } + + /// Number of level-0 data blocks the logical size spans. + pub fn block_count(&self) -> u64 { + blocks_for_size(self.size, self.record_size()) + } + + /// Number of blocks at `level`: level 0 is data, higher levels are the + /// indirect blocks needed to address them. + pub fn blocks_at_level(&self, level: u8) -> u64 { + blocks_at_level(self.block_count(), self.fanout(), level) + } + + /// Tree height required for this object's current size. + pub fn required_levels(&self) -> u8 { + levels_for(self.block_count(), self.fanout()) + } + + pub fn encode(&self) -> FsResult<[u8; DNODE_LEN]> { + if self.inline.len() > INLINE_CAP { + return Err(FsError::Invalid("dnode inline data exceeds capacity")); + } + let mut b = [0u8; DNODE_LEN]; + b[0..8].copy_from_slice(&self.objid.to_be_bytes()); + b[8] = self.kind as u8; + b[9] = self.nlevels; + b[10] = self.record_shift; + b[RESERVED_KIND_PAD] = 0; + b[12..16].copy_from_slice(&self.nlink.to_be_bytes()); + b[16..20].copy_from_slice(&self.mode.to_be_bytes()); + b[20..24].copy_from_slice(&self.uid.to_be_bytes()); + b[24..28].copy_from_slice(&self.gid.to_be_bytes()); + b[28..36].copy_from_slice(&self.size.to_be_bytes()); + b[36..44].copy_from_slice(&self.atime_nanos.to_be_bytes()); + b[44..52].copy_from_slice(&self.mtime_nanos.to_be_bytes()); + b[52..60].copy_from_slice(&self.ctime_nanos.to_be_bytes()); + b[60..68].copy_from_slice(&self.btime_nanos.to_be_bytes()); + b[68..76].copy_from_slice(&self.gen.to_be_bytes()); + b[76..78].copy_from_slice(&(self.inline.len() as u16).to_be_bytes()); + b[PARENT_OFFSET..PARENT_OFFSET + 8].copy_from_slice(&self.parent_objid.to_be_bytes()); + b[BLKPTR_OFFSET..BLKPTR_OFFSET + BLKPTR_LEN].copy_from_slice(&self.blkptr.encode()); + b[INLINE_OFFSET..INLINE_OFFSET + self.inline.len()].copy_from_slice(&self.inline); + Ok(b) + } + + pub fn decode(bytes: &[u8]) -> FsResult { + if bytes.len() != DNODE_LEN { + return Err(FsError::Integrity("dnode: wrong length")); + } + if bytes[RESERVED0].iter().any(|&b| b != 0) + || bytes[RESERVED1].iter().any(|&b| b != 0) + || bytes[RESERVED_KIND_PAD] != 0 + { + return Err(FsError::Integrity("dnode: reserved bytes not zero")); + } + + let inline_len = u16::from_be_bytes(bytes[76..78].try_into().expect("2 bytes")) as usize; + if inline_len > INLINE_CAP { + return Err(FsError::Integrity("dnode: inline length exceeds capacity")); + } + // Bytes past `inline_len` must be zero. Otherwise the tail is a free + // channel for smuggling data through a structure that is otherwise + // fully accounted for. + if bytes[INLINE_OFFSET + inline_len..RESERVED1.start] + .iter() + .any(|&b| b != 0) + { + return Err(FsError::Integrity("dnode: inline padding not zero")); + } + + let record_shift = bytes[10]; + if !(12..=20).contains(&record_shift) { + return Err(FsError::Integrity("dnode: record_shift out of range")); + } + + let d = Dnode { + objid: u64::from_be_bytes(bytes[0..8].try_into().expect("8 bytes")), + kind: DnodeKind::from_u8(bytes[8])?, + nlevels: bytes[9], + record_shift, + parent_objid: u64::from_be_bytes( + bytes[PARENT_OFFSET..PARENT_OFFSET + 8] + .try_into() + .expect("8 bytes"), + ), + nlink: u32::from_be_bytes(bytes[12..16].try_into().expect("4 bytes")), + mode: u32::from_be_bytes(bytes[16..20].try_into().expect("4 bytes")), + uid: u32::from_be_bytes(bytes[20..24].try_into().expect("4 bytes")), + gid: u32::from_be_bytes(bytes[24..28].try_into().expect("4 bytes")), + size: u64::from_be_bytes(bytes[28..36].try_into().expect("8 bytes")), + atime_nanos: u64::from_be_bytes(bytes[36..44].try_into().expect("8 bytes")), + mtime_nanos: u64::from_be_bytes(bytes[44..52].try_into().expect("8 bytes")), + ctime_nanos: u64::from_be_bytes(bytes[52..60].try_into().expect("8 bytes")), + btime_nanos: u64::from_be_bytes(bytes[60..68].try_into().expect("8 bytes")), + gen: u64::from_be_bytes(bytes[68..76].try_into().expect("8 bytes")), + inline: bytes[INLINE_OFFSET..INLINE_OFFSET + inline_len].to_vec(), + blkptr: BlkPtr::decode(&bytes[BLKPTR_OFFSET..BLKPTR_OFFSET + BLKPTR_LEN])?, + }; + + if d.nlevels > MAX_NLEVELS { + return Err(FsError::Integrity("dnode: nlevels exceeds maximum")); + } + // The height and the root pointer must agree. Disagreement would let a + // forged dnode redirect a read to a block at the wrong level, where + // the AAD check is the only thing left standing. + if d.nlevels == 0 { + if !d.blkptr.is_hole() { + return Err(FsError::Integrity("dnode: empty tree with a live pointer")); + } + } else if !d.blkptr.is_hole() && d.blkptr.level != d.nlevels - 1 { + return Err(FsError::Integrity("dnode: root pointer level mismatch")); + } + Ok(d) + } +} + +/// Blocks needed to hold `size` bytes. +pub fn blocks_for_size(size: u64, record_size: usize) -> u64 { + let rs = record_size as u64; + size.div_ceil(rs) +} + +/// Blocks at `level` given `nblocks` data blocks and this fan-out. +pub fn blocks_at_level(nblocks: u64, fanout: u64, level: u8) -> u64 { + let mut n = nblocks; + for _ in 0..level { + n = n.div_ceil(fanout); + } + n +} + +/// Tree height needed to address `nblocks` data blocks. +/// +/// `0` for an empty object, `1` for a single block (the dnode's pointer *is* +/// the data block), and one more level for each factor of `fanout` beyond that. +pub fn levels_for(nblocks: u64, fanout: u64) -> u8 { + if nblocks == 0 { + return 0; + } + let mut levels: u8 = 1; + let mut capacity: u64 = 1; + while capacity < nblocks { + capacity = capacity.saturating_mul(fanout); + levels += 1; + } + levels +} + +/// Index of the entry within a level-`level` indirect block that leads toward +/// data block `block_index`. +pub fn entry_index(block_index: u64, level: u8, fanout: u64) -> u64 { + debug_assert!(level >= 1, "level 0 blocks have no entries"); + (block_index / fanout.pow(u32::from(level) - 1)) % fanout +} + +/// Index, within its own level, of the level-`level` block covering data block +/// `block_index`. This is the value fed to the AEAD as `block_index`, so it +/// must be identical on the write and read paths. +pub fn block_index_at_level(block_index: u64, level: u8, fanout: u64) -> u64 { + block_index / fanout.pow(u32::from(level)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::crypto::Hash256; + use crate::store::blkptr::{Compression, Dva}; + + const FANOUT: u64 = 1024; + + fn ptr(level: u8) -> BlkPtr { + BlkPtr { + dva: Dva { + txg: 5, + slab: 1, + offset: 64, + len: 1024 + 16, + }, + logical_len: 1024, + level, + compression: Compression::None, + fill: 1, + birth_txg: 5, + nonce: crate::crypto::BlockNonce::new(5, 2), + checksum: Hash256::of(b"block"), + } + } + + fn sample() -> Dnode { + Dnode { + objid: 42, + kind: DnodeKind::File, + nlevels: 2, + record_shift: 17, + parent_objid: 1, + nlink: 1, + mode: 0o100644, + uid: 1000, + gid: 1000, + size: 300_000, + atime_nanos: 111, + mtime_nanos: 222, + ctime_nanos: 333, + btime_nanos: 444, + gen: 7, + inline: Vec::new(), + blkptr: ptr(1), + } + } + + #[test] + fn dnode_divides_the_record_evenly() { + assert_eq!(DNODE_LEN, 512); + assert_eq!(dnodes_per_block(128 * 1024), 256); + assert_eq!(128 * 1024 % DNODE_LEN, 0); + assert_eq!(INLINE_OFFSET + INLINE_CAP, RESERVED1.start); + assert_eq!(BLKPTR_OFFSET, PARENT_OFFSET + 8); + } + + #[test] + fn round_trips() { + let d = sample(); + assert_eq!(Dnode::decode(&d.encode().unwrap()).unwrap(), d); + } + + #[test] + fn free_dnode_round_trips() { + let d = Dnode::free(9, 17); + let decoded = Dnode::decode(&d.encode().unwrap()).unwrap(); + assert_eq!(decoded, d); + assert_eq!(decoded.kind, DnodeKind::Free); + assert!(decoded.blkptr.is_hole()); + } + + #[test] + fn inline_data_round_trips() { + let mut d = Dnode::new(3, DnodeKind::Symlink, 17, 99); + d.inline = b"../relative/target/path".to_vec(); + d.size = d.inline.len() as u64; + let decoded = Dnode::decode(&d.encode().unwrap()).unwrap(); + assert_eq!(decoded.inline, b"../relative/target/path"); + } + + #[test] + fn inline_at_exact_capacity_round_trips() { + let mut d = Dnode::new(3, DnodeKind::Symlink, 17, 0); + d.inline = vec![0x41; INLINE_CAP]; + let decoded = Dnode::decode(&d.encode().unwrap()).unwrap(); + assert_eq!(decoded.inline.len(), INLINE_CAP); + } + + #[test] + fn inline_beyond_capacity_is_rejected_at_encode() { + let mut d = Dnode::new(3, DnodeKind::Symlink, 17, 0); + d.inline = vec![0x41; INLINE_CAP + 1]; + assert!(d.encode().is_err()); + } + + #[test] + fn field_offsets_are_stable() { + let b = sample().encode().unwrap(); + assert_eq!(&b[0..8], &42u64.to_be_bytes()); + assert_eq!(b[8], DnodeKind::File as u8); + assert_eq!(b[9], 2); + assert_eq!(b[10], 17); + assert_eq!(b[RESERVED_KIND_PAD], 0); + assert_eq!( + &b[PARENT_OFFSET..PARENT_OFFSET + 8], + &1u64.to_be_bytes(), + "parent link must sit immediately before the block pointer" + ); + assert_eq!(&b[28..36], &300_000u64.to_be_bytes()); + assert_eq!(&b[68..76], &7u64.to_be_bytes()); + assert_eq!( + &b[BLKPTR_OFFSET..BLKPTR_OFFSET + BLKPTR_LEN], + &ptr(1).encode() + ); + } + + #[test] + fn rejects_wrong_length() { + assert!(Dnode::decode(&[0u8; 511]).is_err()); + assert!(Dnode::decode(&[0u8; 513]).is_err()); + } + + #[test] + fn rejects_unknown_kind() { + let mut b = sample().encode().unwrap(); + b[8] = 99; + assert!(matches!( + Dnode::decode(&b), + Err(FsError::Integrity("dnode: unknown kind")) + )); + } + + #[test] + fn rejects_reserved_bytes() { + for i in [78, 79, 510, 511] { + let mut b = sample().encode().unwrap(); + b[i] = 1; + assert!(Dnode::decode(&b).is_err(), "byte {i} was ignored"); + } + } + + /// A non-zero tail past `inline_len` would be an unaccounted channel in an + /// otherwise fully-described structure. + #[test] + fn rejects_dirty_inline_padding() { + let mut d = Dnode::new(3, DnodeKind::Symlink, 17, 0); + d.inline = b"short".to_vec(); + let mut b = d.encode().unwrap(); + b[INLINE_OFFSET + 10] = 0xff; + assert!(matches!( + Dnode::decode(&b), + Err(FsError::Integrity("dnode: inline padding not zero")) + )); + } + + #[test] + fn rejects_out_of_range_record_shift() { + for shift in [11u8, 21, 63] { + let mut b = sample().encode().unwrap(); + b[10] = shift; + assert!(Dnode::decode(&b).is_err(), "accepted shift {shift}"); + } + } + + /// The height and the root pointer must agree, or a forged dnode could + /// redirect a read to a block at a level it was never sealed for. + #[test] + fn rejects_level_disagreement() { + let mut d = sample(); + d.nlevels = 3; // pointer is still level 1 + assert!(matches!( + Dnode::decode(&d.encode().unwrap()), + Err(FsError::Integrity("dnode: root pointer level mismatch")) + )); + + let mut d = sample(); + d.nlevels = 0; // but the pointer is live + assert!(matches!( + Dnode::decode(&d.encode().unwrap()), + Err(FsError::Integrity("dnode: empty tree with a live pointer")) + )); + } + + #[test] + fn rejects_excessive_nlevels() { + let mut d = sample(); + d.nlevels = MAX_NLEVELS + 1; + d.blkptr = BlkPtr::HOLE; + assert!(Dnode::decode(&d.encode().unwrap()).is_err()); + } + + // ---- tree geometry ----------------------------------------------------- + + #[test] + fn blocks_for_size_rounds_up() { + assert_eq!(blocks_for_size(0, 4096), 0); + assert_eq!(blocks_for_size(1, 4096), 1); + assert_eq!(blocks_for_size(4096, 4096), 1); + assert_eq!(blocks_for_size(4097, 4096), 2); + } + + #[test] + fn levels_grow_with_the_block_count() { + assert_eq!(levels_for(0, FANOUT), 0); + assert_eq!(levels_for(1, FANOUT), 1); + assert_eq!(levels_for(2, FANOUT), 2); + assert_eq!(levels_for(FANOUT, FANOUT), 2); + assert_eq!(levels_for(FANOUT + 1, FANOUT), 3); + assert_eq!(levels_for(FANOUT * FANOUT, FANOUT), 3); + assert_eq!(levels_for(FANOUT * FANOUT + 1, FANOUT), 4); + } + + /// Height must stay within what a block pointer can encode, for every + /// permitted record size at the largest file `u64` bytes can express. + /// + /// The worst case is the *smallest* record: fewer pointers per indirect + /// block means a taller tree. This is what sets `MAX_LEVEL`. + #[test] + fn levels_stay_within_the_pointer_limit_at_every_record_size() { + for shift in 12u8..=20 { + let record = 1usize << shift; + let fanout = (record / BLKPTR_LEN) as u64; + let max_blocks = blocks_for_size(u64::MAX, record); + let levels = levels_for(max_blocks, fanout); + assert!( + levels <= MAX_NLEVELS, + "record {record}: {levels} levels exceeds {MAX_NLEVELS}" + ); + } + } + + /// The smallest record size is the binding constraint, and it is close + /// enough to the limit to be worth pinning exactly — if `MAX_LEVEL` is + /// ever lowered, this fails rather than silently rejecting large files. + #[test] + fn worst_case_height_is_the_minimum_record_size() { + let fanout = (4096 / BLKPTR_LEN) as u64; + assert_eq!(fanout, 32); + assert_eq!(levels_for(blocks_for_size(u64::MAX, 4096), fanout), 12); + assert_eq!( + levels_for(blocks_for_size(u64::MAX, 128 * 1024), FANOUT), + 6, + "the default record size is nowhere near the limit" + ); + } + + #[test] + fn blocks_at_level_collapses_toward_one() { + assert_eq!(blocks_at_level(2049, FANOUT, 0), 2049); + assert_eq!(blocks_at_level(2049, FANOUT, 1), 3); + assert_eq!(blocks_at_level(2049, FANOUT, 2), 1); + assert_eq!(blocks_at_level(0, FANOUT, 1), 0); + } + + #[test] + fn entry_and_block_indices_are_consistent() { + // Data block 1025 with fan-out 1024 lives in level-1 block 1, entry 1. + assert_eq!(entry_index(1025, 1, FANOUT), 1); + assert_eq!(block_index_at_level(1025, 1, FANOUT), 1); + // That level-1 block is entry 1 of the single level-2 block. + assert_eq!(entry_index(1025, 2, FANOUT), 1); + assert_eq!(block_index_at_level(1025, 2, FANOUT), 0); + } + + #[test] + fn level_zero_index_is_the_block_itself() { + for i in [0u64, 1, 1023, 1024, 1_000_000] { + assert_eq!(block_index_at_level(i, 0, FANOUT), i); + } + } + + /// Walking the entry indices from the root down must reconstruct the + /// original data block index. This is the invariant the tree walker + /// depends on, in both directions. + #[test] + fn entry_path_reconstructs_the_block_index() { + let fanout = 4u64; + for block in 0..64u64 { + let levels = levels_for(block + 1, fanout); + let mut rebuilt = 0u64; + for level in (1..levels).rev() { + rebuilt = rebuilt * fanout + entry_index(block, level, fanout); + } + assert_eq!(rebuilt, block, "path for block {block} did not round-trip"); + } + } +} diff --git a/crates/s3fs-core/src/store/indirect.rs b/crates/s3fs-core/src/store/indirect.rs new file mode 100644 index 0000000..4b02934 --- /dev/null +++ b/crates/s3fs-core/src/store/indirect.rs @@ -0,0 +1,792 @@ +//! The indirect block tree: reading down it, and rebuilding it copy-on-write. +//! +//! An object's data blocks hang off a tree of indirect blocks, each of which +//! is an array of [`BlkPtr`]s. The dnode holds the single pointer at the top. +//! Nothing is ever modified in place: a write produces new blocks bottom-up, +//! new indirect blocks above them, and finally a new root pointer. +//! +//! ## Indirect blocks are variable length +//! +//! An indirect block stores only as many pointers as it actually needs, and +//! any entry past the stored length reads as a hole. Without this, a two-block +//! file would pay a full 128 KiB indirect block to hold two used pointers and +//! 1022 zeros — a 50% storage overhead on a 256 KiB file. Trailing holes are +//! trimmed on write for the same reason, which also makes sparse files cheap +//! at every level of the tree rather than only at the leaves. +//! +//! ## What the caller must dirty +//! +//! The rebuild rewrites exactly the blocks it is given plus the indirect path +//! above them, and — when the object's size changed — the boundary block at +//! each level, so that pointers past the new end are dropped rather than left +//! dangling. + +use std::collections::{BTreeMap, BTreeSet}; + +use bytes::Bytes; + +use crate::errors::{FsError, FsResult}; + +use super::blkptr::{BlkPtr, BLKPTR_LEN}; +use super::blockstore::BlockStore; +use super::dnode::{blocks_at_level, blocks_for_size, levels_for, Dnode}; +use super::slab::SlabWriter; + +/// Decode entry `slot` of an indirect block's contents. +/// +/// Slots past the stored length are holes — that is what makes indirect blocks +/// variable length. +fn entry_at(block: &[u8], slot: u64) -> FsResult { + let range = usize::try_from(slot) + .ok() + .and_then(|s| s.checked_mul(BLKPTR_LEN)) + .and_then(|start| Some(start..start.checked_add(BLKPTR_LEN)?)); + match range.and_then(|r| block.get(r)) { + Some(raw) => BlkPtr::decode(raw), + None => Ok(BlkPtr::HOLE), + } +} + +/// Decode a whole indirect block into its pointer array. +fn decode_entries(block: &[u8]) -> FsResult> { + if !block.len().is_multiple_of(BLKPTR_LEN) { + return Err(FsError::Integrity( + "indirect block length is not a multiple of the pointer size", + )); + } + let (entries, _) = block.as_chunks::(); + entries.iter().map(|e| BlkPtr::decode(e)).collect() +} + +fn encode_entries(entries: &[BlkPtr]) -> Vec { + let mut out = Vec::with_capacity(entries.len() * BLKPTR_LEN); + for e in entries { + out.extend_from_slice(&e.encode()); + } + out +} + +/// Logical length of data block `index` given the object's size. +/// +/// The last block is short; blocks past the end are empty. +pub fn block_logical_len(size: u64, record_size: usize, index: u64) -> usize { + let rs = record_size as u64; + let start = index.saturating_mul(rs); + if start >= size { + return 0; + } + (size - start).min(rs) as usize +} + +/// Resolve the pointer stored at `(level, index)` of an object's tree. +/// +/// Walks down from the dnode's root pointer, verifying every block on the way. +/// Returns a hole for anything not present — a sparse region, a level above +/// the current tree height, or an index past the end. +pub async fn resolve_ptr( + bs: &BlockStore, + dnode: &Dnode, + level: u8, + index: u64, +) -> FsResult { + if dnode.nlevels == 0 { + return Ok(BlkPtr::HOLE); + } + let root_level = dnode.nlevels - 1; + if level > root_level { + return Ok(BlkPtr::HOLE); + } + let fanout = dnode.fanout(); + + let mut ptr = dnode.blkptr; + let mut cur_level = root_level; + // Index of the block we are currently sitting on, within its own level. + let mut cur_index = index / pow(fanout, u32::from(cur_level - level))?; + if cur_index != 0 { + // The root is the only block at its level, so an index that maps above + // it lies beyond what this tree currently covers. That is an ordinary + // question during a rebuild — "does the old tree already have a block + // here?" — and the answer is simply no. + return Ok(BlkPtr::HOLE); + } + + while cur_level > level { + if ptr.is_hole() { + return Ok(BlkPtr::HOLE); + } + let block = bs.read_block(&ptr, dnode.objid, cur_index).await?; + let slot = (index / pow(fanout, u32::from(cur_level - 1 - level))?) % fanout; + ptr = entry_at(&block, slot)?; + cur_level -= 1; + cur_index = index / pow(fanout, u32::from(cur_level - level))?; + } + Ok(ptr) +} + +fn pow(fanout: u64, exp: u32) -> FsResult { + fanout + .checked_pow(exp) + .ok_or(FsError::Invalid("tree index arithmetic overflowed")) +} + +/// Read data block `index`, returning exactly the bytes the object's size says +/// that block holds. +/// +/// A stored block shorter or longer than the size implies is padded or clipped +/// rather than rejected: `set_size` can change a block's logical length without +/// changing its contents, and both directions are legitimate. +pub async fn read_data_block(bs: &BlockStore, dnode: &Dnode, index: u64) -> FsResult { + let want = block_logical_len(dnode.size, dnode.record_size(), index); + if want == 0 { + return Ok(Bytes::new()); + } + let ptr = resolve_ptr(bs, dnode, 0, index).await?; + if ptr.is_hole() { + return Ok(Bytes::from(vec![0u8; want])); + } + let got = bs.read_block(&ptr, dnode.objid, index).await?; + Ok(fit(got, want)) +} + +/// Read block `index` exactly as it was stored, with no size-based padding or +/// clipping. +/// +/// [`read_data_block`] fits its result to what the object's `size` implies, +/// which is right for files — a file is a byte range, and the tail block is +/// short. It is wrong for objects whose blocks are self-describing structures +/// of their own length, such as directory blocks, where padding a 200-byte +/// leaf out to a full record would be pure waste. +/// +/// `None` means the block is a hole. +pub async fn read_raw_block(bs: &BlockStore, dnode: &Dnode, index: u64) -> FsResult> { + let ptr = resolve_ptr(bs, dnode, 0, index).await?; + if ptr.is_hole() { + return Ok(None); + } + Ok(Some(bs.read_block(&ptr, dnode.objid, index).await?)) +} + +fn fit(block: Bytes, want: usize) -> Bytes { + match block.len().cmp(&want) { + std::cmp::Ordering::Equal => block, + std::cmp::Ordering::Greater => block.slice(..want), + std::cmp::Ordering::Less => { + let mut v = block.to_vec(); + v.resize(want, 0); + Bytes::from(v) + } + } +} + +/// Rebuild an object's tree with `dirty` level-0 blocks applied and the object +/// resized to `new_size`. +/// +/// Returns the updated dnode. Nothing is written to the backend here — blocks +/// are staged in `writer`, and become durable when the transaction group's +/// slabs are PUT. +pub async fn commit_object( + bs: &BlockStore, + writer: &mut SlabWriter, + dnode: &Dnode, + dirty: BTreeMap>, + new_size: u64, +) -> FsResult { + let record_size = dnode.record_size(); + let fanout = dnode.fanout(); + let old_nblocks = dnode.block_count(); + let new_nblocks = blocks_for_size(new_size, record_size); + let new_levels = levels_for(new_nblocks, fanout); + let size_changed = new_nblocks != old_nblocks; + + // If the object shrank enough to lose levels, the new root is an interior + // node of the old tree. Resolve it now so the rebuild below sees a tree of + // the right height. + let (base_ptr, base_levels) = if new_levels < dnode.nlevels { + if new_levels == 0 { + (BlkPtr::HOLE, 0) + } else { + (resolve_ptr(bs, dnode, new_levels - 1, 0).await?, new_levels) + } + } else { + (dnode.blkptr, dnode.nlevels) + }; + + // Level 0: stage every dirty block that survives the resize. + let mut current: BTreeMap = BTreeMap::new(); + for (index, data) in dirty { + if index >= new_nblocks { + continue; // written and then truncated away in the same txg + } + let ptr = writer.write_block(bs.keys(), dnode.objid, 0, index, 1, &data)?; + current.insert(index, ptr); + } + + for level in 1..new_levels { + // When the tree gains height, the previous root becomes entry 0 of its + // own level in the new tree. It is not dirty, so nothing else would + // carry it upward. + if base_levels >= 1 && level == base_levels && new_levels > base_levels { + current.entry(0).or_insert(base_ptr); + } + + let child_count = blocks_at_level(new_nblocks, fanout, level - 1); + let mut parents: BTreeSet = current.keys().map(|i| i / fanout).collect(); + // A resize moves the end of every level, so the block containing the + // new last child must be rewritten to drop what is now past the end. + if size_changed && child_count > 0 { + parents.insert((child_count - 1) / fanout); + } + + let mut next: BTreeMap = BTreeMap::new(); + for parent in parents { + let first_child = parent.saturating_mul(fanout); + if first_child >= child_count { + continue; // entirely past the new end; simply unreferenced + } + let valid = (child_count - first_child).min(fanout) as usize; + + let mut entries = load_entries(bs, dnode, level, parent, base_levels).await?; + entries.resize(valid, BlkPtr::HOLE); + for (index, ptr) in current.range(first_child..first_child + fanout) { + entries[(index - first_child) as usize] = *ptr; + } + while entries.last().is_some_and(|e| e.is_hole()) { + entries.pop(); + } + + let ptr = if entries.is_empty() { + BlkPtr::HOLE + } else { + let fill = entries.iter().fold(0u32, |a, e| a.saturating_add(e.fill)); + writer.write_block( + bs.keys(), + dnode.objid, + level, + parent, + fill, + &encode_entries(&entries), + )? + }; + next.insert(parent, ptr); + } + current = next; + } + + let root = match new_levels { + 0 => BlkPtr::HOLE, + _ => match current.remove(&0) { + Some(p) => p, + // Nothing under the root changed, so the old root still stands. + None if base_levels == new_levels => base_ptr, + None => BlkPtr::HOLE, + }, + }; + + Ok(Dnode { + size: new_size, + nlevels: if root.is_hole() { 0 } else { new_levels }, + blkptr: root, + ..dnode.clone() + }) +} + +/// Read the existing entries of the level-`level` block at `index`. +/// +/// Empty when that block does not exist yet, which is the case for every level +/// above the old tree's height. +async fn load_entries( + bs: &BlockStore, + dnode: &Dnode, + level: u8, + index: u64, + base_levels: u8, +) -> FsResult> { + if base_levels == 0 || level > base_levels - 1 { + return Ok(Vec::new()); + } + let ptr = resolve_ptr(bs, dnode, level, index).await?; + if ptr.is_hole() { + return Ok(Vec::new()); + } + let block = bs.read_block(&ptr, dnode.objid, index).await?; + decode_entries(&block) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::backend::memory::MemoryBackend; + use crate::backend::{Backend, PutBlobInput}; + use crate::crypto::{KeyMaterial, MasterSecret}; + use crate::store::config::StoreConfig; + use crate::store::dnode::DnodeKind; + use std::sync::Arc; + + /// Small records so multi-level trees are reachable in a test: 4 KiB + /// records give a fan-out of 32, so three levels cover 32 768 blocks. + fn small_config() -> StoreConfig { + StoreConfig { + record_size: 4096, + slab_max_bytes: 16 * 1024 * 1024, + ..Default::default() + } + } + + fn store(config: StoreConfig) -> (Arc, BlockStore) { + let backend = Arc::new(MemoryBackend::new()); + let keys = + Arc::new(KeyMaterial::derive(&MasterSecret::from_bytes([2u8; 32]), [0u8; 16]).unwrap()); + let bs = BlockStore::new(backend.clone(), keys, Arc::new(config)); + (backend, bs) + } + + fn empty_file(record_shift: u8) -> Dnode { + Dnode::new(42, DnodeKind::File, record_shift, 0) + } + + /// Apply one transaction group: stage the dirty blocks, rebuild the tree, + /// PUT the slabs, and return the new dnode. + async fn commit( + bs: &BlockStore, + txg: u64, + dnode: &Dnode, + dirty: BTreeMap>, + new_size: u64, + ) -> Dnode { + let mut w = SlabWriter::new(txg, bs.config()); + let out = commit_object(bs, &mut w, dnode, dirty, new_size) + .await + .unwrap(); + bs.write_slabs(txg, w.finish().unwrap()).await.unwrap(); + out + } + + fn blocks(spec: &[(u64, u8, usize)]) -> BTreeMap> { + spec.iter() + .map(|&(idx, fill, len)| (idx, vec![fill; len])) + .collect() + } + + async fn read_all(bs: &BlockStore, d: &Dnode) -> Vec { + let mut out = Vec::new(); + for i in 0..d.block_count() { + out.extend_from_slice(&read_data_block(bs, d, i).await.unwrap()); + } + out + } + + #[tokio::test] + async fn empty_object_has_no_tree() { + let (backend, bs) = store(small_config()); + let d = commit(&bs, 1, &empty_file(12), BTreeMap::new(), 0).await; + + assert_eq!(d.nlevels, 0); + assert!(d.blkptr.is_hole()); + assert_eq!(d.size, 0); + assert_eq!(backend.object_count(), 0); + } + + #[tokio::test] + async fn single_block_needs_no_indirect_block() { + let (_b, bs) = store(small_config()); + let d = commit(&bs, 1, &empty_file(12), blocks(&[(0, 0xaa, 4096)]), 4096).await; + + assert_eq!(d.nlevels, 1, "the dnode pointer is the data block itself"); + assert_eq!(d.blkptr.level, 0); + assert_eq!(read_data_block(&bs, &d, 0).await.unwrap(), vec![0xaa; 4096]); + } + + #[tokio::test] + async fn short_tail_block_reads_at_its_real_length() { + let (_b, bs) = store(small_config()); + let d = commit(&bs, 1, &empty_file(12), blocks(&[(0, 7, 100)]), 100).await; + assert_eq!(read_data_block(&bs, &d, 0).await.unwrap(), vec![7u8; 100]); + assert_eq!(d.block_count(), 1); + } + + #[tokio::test] + async fn two_blocks_add_one_level() { + let (_b, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 1, 4096), (1, 2, 4096)]), + 8192, + ) + .await; + + assert_eq!(d.nlevels, 2); + assert_eq!(d.blkptr.level, 1); + assert_eq!(read_data_block(&bs, &d, 0).await.unwrap(), vec![1u8; 4096]); + assert_eq!(read_data_block(&bs, &d, 1).await.unwrap(), vec![2u8; 4096]); + } + + /// The economy that variable-length indirect blocks buy: a two-block file + /// must not pay for 32 pointer slots. + #[tokio::test] + async fn indirect_blocks_hold_only_the_pointers_they_need() { + let (_b, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 1, 4096), (1, 2, 4096)]), + 8192, + ) + .await; + assert_eq!( + d.blkptr.logical_len as usize, + 2 * BLKPTR_LEN, + "indirect block should hold exactly two pointers" + ); + } + + #[tokio::test] + async fn three_levels() { + let (_b, bs) = store(small_config()); + // Fan-out 32, so block 1000 needs three levels. + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(1000, 9, 4096)]), + 1001 * 4096, + ) + .await; + + assert_eq!(d.nlevels, 3); + assert_eq!(d.blkptr.level, 2); + assert_eq!( + read_data_block(&bs, &d, 1000).await.unwrap(), + vec![9u8; 4096] + ); + } + + #[tokio::test] + async fn sparse_regions_read_as_zeros() { + let (backend, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(100, 5, 4096)]), + 101 * 4096, + ) + .await; + + assert_eq!( + read_data_block(&bs, &d, 100).await.unwrap(), + vec![5u8; 4096] + ); + for i in [0u64, 1, 50, 99] { + assert_eq!( + read_data_block(&bs, &d, i).await.unwrap(), + vec![0u8; 4096], + "block {i} should be a hole" + ); + } + // One data block plus its indirect path — nowhere near 101 blocks. + assert!(backend.object_count() <= 2); + } + + #[tokio::test] + async fn fill_counts_live_blocks_beneath_a_pointer() { + let (_b, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 1, 4096), (5, 1, 4096), (31, 1, 4096)]), + 32 * 4096, + ) + .await; + assert_eq!(d.blkptr.fill, 3); + } + + // ---- copy-on-write across transaction groups --------------------------- + + #[tokio::test] + async fn rewriting_one_block_preserves_the_others() { + let (_b, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 1, 4096), (1, 2, 4096), (2, 3, 4096)]), + 3 * 4096, + ) + .await; + + let d = commit(&bs, 2, &d, blocks(&[(1, 0xff, 4096)]), 3 * 4096).await; + + assert_eq!(read_data_block(&bs, &d, 0).await.unwrap(), vec![1u8; 4096]); + assert_eq!(read_data_block(&bs, &d, 1).await.unwrap(), vec![0xff; 4096]); + assert_eq!(read_data_block(&bs, &d, 2).await.unwrap(), vec![3u8; 4096]); + } + + /// The old tree must still be readable after a new one is committed — + /// that is what makes an old root a usable snapshot. + #[tokio::test] + async fn the_previous_tree_remains_readable() { + let (_b, bs) = store(small_config()); + let old = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 1, 4096), (1, 2, 4096)]), + 2 * 4096, + ) + .await; + let new = commit(&bs, 2, &old, blocks(&[(0, 0xee, 4096)]), 2 * 4096).await; + + assert_eq!( + read_data_block(&bs, &new, 0).await.unwrap(), + vec![0xee; 4096] + ); + assert_eq!( + read_data_block(&bs, &old, 0).await.unwrap(), + vec![1u8; 4096], + "the superseded tree must be unchanged" + ); + assert_ne!(old.blkptr.dva, new.blkptr.dva); + } + + #[tokio::test] + async fn appending_grows_the_tree_and_keeps_existing_data() { + let (_b, bs) = store(small_config()); + // One block, height 1. + let d = commit(&bs, 1, &empty_file(12), blocks(&[(0, 1, 4096)]), 4096).await; + assert_eq!(d.nlevels, 1); + + // Append a second block: height must grow to 2 and block 0 must + // survive even though it was never dirtied. + let d = commit(&bs, 2, &d, blocks(&[(1, 2, 4096)]), 2 * 4096).await; + assert_eq!(d.nlevels, 2); + assert_eq!(read_data_block(&bs, &d, 0).await.unwrap(), vec![1u8; 4096]); + assert_eq!(read_data_block(&bs, &d, 1).await.unwrap(), vec![2u8; 4096]); + + // And again, across a second height increase. + let d = commit(&bs, 3, &d, blocks(&[(40, 3, 4096)]), 41 * 4096).await; + assert_eq!(d.nlevels, 3); + assert_eq!(read_data_block(&bs, &d, 0).await.unwrap(), vec![1u8; 4096]); + assert_eq!(read_data_block(&bs, &d, 1).await.unwrap(), vec![2u8; 4096]); + assert_eq!(read_data_block(&bs, &d, 40).await.unwrap(), vec![3u8; 4096]); + } + + #[tokio::test] + async fn growing_by_many_levels_at_once() { + let (_b, bs) = store(small_config()); + let d = commit(&bs, 1, &empty_file(12), blocks(&[(0, 1, 4096)]), 4096).await; + // Fan-out 32, so 40 001 blocks needs 32^4 = 1 048 576 of capacity: + // one jump from height 1 straight to height 5. + let d = commit(&bs, 2, &d, blocks(&[(40_000, 2, 4096)]), 40_001 * 4096).await; + + assert_eq!(d.nlevels, 5); + assert_eq!(read_data_block(&bs, &d, 0).await.unwrap(), vec![1u8; 4096]); + assert_eq!( + read_data_block(&bs, &d, 40_000).await.unwrap(), + vec![2u8; 4096] + ); + } + + #[tokio::test] + async fn truncating_shrinks_the_tree() { + let (_b, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 4, 4096), (500, 7, 4096)]), + 501 * 4096, + ) + .await; + assert_eq!(d.nlevels, 3); + + let d = commit(&bs, 2, &d, BTreeMap::new(), 4096).await; + assert_eq!(d.nlevels, 1, "one block needs no indirect blocks"); + assert_eq!(d.size, 4096); + assert_eq!(d.block_count(), 1); + assert_eq!( + read_data_block(&bs, &d, 0).await.unwrap(), + vec![4u8; 4096], + "the surviving block keeps its contents through the height change" + ); + } + + /// Truncating a sparse file down to a region that holds no data at all + /// leaves a live object with no tree — every read is a hole. The size is + /// still 4 KiB; there is simply nothing stored behind it. + #[tokio::test] + async fn truncating_onto_a_hole_leaves_an_empty_tree() { + let (_b, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(500, 7, 4096)]), + 501 * 4096, + ) + .await; + let d = commit(&bs, 2, &d, BTreeMap::new(), 4096).await; + + assert_eq!(d.nlevels, 0); + assert!(d.blkptr.is_hole()); + assert_eq!(d.size, 4096); + assert_eq!(read_data_block(&bs, &d, 0).await.unwrap(), vec![0u8; 4096]); + } + + #[tokio::test] + async fn truncating_to_zero_empties_the_tree() { + let (_b, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 1, 4096), (1, 2, 4096)]), + 2 * 4096, + ) + .await; + let d = commit(&bs, 2, &d, BTreeMap::new(), 0).await; + + assert_eq!(d.nlevels, 0); + assert!(d.blkptr.is_hole()); + assert_eq!(d.size, 0); + } + + #[tokio::test] + async fn truncating_drops_blocks_past_the_new_end() { + let (_b, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 1, 4096), (1, 2, 4096), (2, 3, 4096)]), + 3 * 4096, + ) + .await; + let d = commit(&bs, 2, &d, BTreeMap::new(), 2 * 4096).await; + + assert_eq!(d.block_count(), 2); + assert_eq!(read_all(&bs, &d).await.len(), 2 * 4096); + // The pointer array must have shrunk, not merely been ignored. + assert_eq!(d.blkptr.logical_len as usize, 2 * BLKPTR_LEN); + } + + #[tokio::test] + async fn truncate_then_regrow_reads_zeros_not_stale_data() { + let (_b, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 1, 4096), (1, 0xcc, 4096)]), + 2 * 4096, + ) + .await; + let d = commit(&bs, 2, &d, BTreeMap::new(), 4096).await; + let d = commit(&bs, 3, &d, BTreeMap::new(), 2 * 4096).await; + + assert_eq!( + read_data_block(&bs, &d, 1).await.unwrap(), + vec![0u8; 4096], + "regrown region must not resurrect the old block" + ); + } + + #[tokio::test] + async fn a_committed_block_is_written_only_once() { + let (_b, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 1, 4096), (1, 2, 4096)]), + 2 * 4096, + ) + .await; + + // A no-op commit must not rewrite anything: the root pointer stays put. + let same = commit(&bs, 2, &d, BTreeMap::new(), 2 * 4096).await; + assert_eq!( + same.blkptr, d.blkptr, + "an empty txg must not touch the tree" + ); + } + + #[tokio::test] + async fn many_blocks_round_trip() { + let (_b, bs) = store(small_config()); + let spec: Vec<_> = (0..200u64).map(|i| (i, (i % 251) as u8, 4096)).collect(); + let d = commit(&bs, 1, &empty_file(12), blocks(&spec), 200 * 4096).await; + + assert_eq!(d.nlevels, 3); + for i in 0..200u64 { + assert_eq!( + read_data_block(&bs, &d, i).await.unwrap(), + vec![(i % 251) as u8; 4096], + "block {i}" + ); + } + } + + // ---- verification ------------------------------------------------------ + + /// An indirect block is a block like any other: substituting one must fail + /// the checksum in its parent. + #[tokio::test] + async fn a_tampered_indirect_block_is_rejected() { + let (backend, bs) = store(small_config()); + let d = commit( + &bs, + 1, + &empty_file(12), + blocks(&[(0, 1, 4096), (1, 2, 4096)]), + 2 * 4096, + ) + .await; + + let key = bs.config().slab_key(1, 0); + let mut body = backend.get_blob(&key, None).await.unwrap().body.to_vec(); + // The indirect block is written after the data blocks, so corrupt the tail. + let last = body.len() - 1; + body[last] ^= 0xff; + backend + .put_blob(PutBlobInput::new(key, Bytes::from(body))) + .await + .unwrap(); + + assert!(matches!( + read_data_block(&bs, &d, 0).await, + Err(FsError::Integrity(_)) + )); + } + + #[test] + fn entries_past_the_stored_length_are_holes() { + let block = [0u8; BLKPTR_LEN]; + assert!(entry_at(&block, 0).unwrap().is_hole()); + assert!(entry_at(&block, 1).unwrap().is_hole()); + assert!(entry_at(&[], 0).unwrap().is_hole()); + assert!(entry_at(&block, u64::MAX).unwrap().is_hole()); + } + + #[test] + fn ragged_indirect_block_is_rejected() { + assert!(matches!( + decode_entries(&[0u8; BLKPTR_LEN + 1]), + Err(FsError::Integrity(_)) + )); + assert!(decode_entries(&[0u8; BLKPTR_LEN * 2]).is_ok()); + } + + #[test] + fn block_logical_len_handles_the_tail() { + assert_eq!(block_logical_len(0, 4096, 0), 0); + assert_eq!(block_logical_len(100, 4096, 0), 100); + assert_eq!(block_logical_len(4096, 4096, 0), 4096); + assert_eq!(block_logical_len(4097, 4096, 0), 4096); + assert_eq!(block_logical_len(4097, 4096, 1), 1); + assert_eq!(block_logical_len(4097, 4096, 2), 0); + } +} diff --git a/crates/s3fs-core/src/store/mod.rs b/crates/s3fs-core/src/store/mod.rs new file mode 100644 index 0000000..5a1c5ad --- /dev/null +++ b/crates/s3fs-core/src/store/mod.rs @@ -0,0 +1,73 @@ +//! The block store: a ZFS-style copy-on-write object tree over S3. +//! +//! A write allocates new blocks, rebuilds the indirect blocks above them, and +//! ends by publishing a new *root record* whose Merkle root covers the entire +//! filesystem. Old blocks stay exactly where they were, which is what makes +//! every historical root a usable snapshot. +//! +//! ```text +//! roots/ ← signed, hash-chained, Object Lock COMPLIANCE +//! │ meta_dnode BlkPtr ← the Merkle root +//! ▼ +//! meta-dnode ──▶ indirect blocks ──▶ dnodes (one per file/dir/symlink) +//! │ blkptrs +//! ▼ +//! indirect blocks ──▶ data blocks +//! │ +//! ▼ +//! slabs// (packed per txg, unlocked) +//! ``` +//! +//! Each arrow is a [`BlkPtr`] carrying the BLAKE3 checksum of the block it +//! points at, so verifying the root's signature transitively verifies every +//! byte below it. +//! +//! ## Two different senses of "immutable" +//! +//! These are easy to conflate and mean very different things: +//! +//! **The writer never overwrites.** Copy-on-write plus a txg number that is +//! consumed once and never reused means every block lands at an address +//! nothing has occupied before. This is a property of our code, and it holds +//! for slabs and roots alike. It is also what makes Object Lock usable at all: +//! a mutable object could not live in a COMPLIANCE bucket and keep being +//! updated. +//! +//! **S3 prevents *others* from overwriting — only the roots.** Retention is +//! applied to `roots/` and nothing else. That is the anchor, and it is +//! the whole of the rollback guarantee. Slabs deliberately live in an unlocked +//! bucket so that dead copy-on-write blocks stay reclaimable, which means +//! anyone with write access to the data bucket can delete one. Doing so is a +//! *detectable denial of service* — the read fails with +//! [`crate::errors::FsError::Integrity`] and the root chain still verifies — +//! never a rollback and never a forgery. +//! +//! One object is genuinely mutable by design: `roots/latest`, the tip hint. It +//! is rewritten on every commit, is never trusted, and exists only to turn tip +//! discovery into one HEAD instead of a search. The mount protocol verifies it +//! by probing for `seq + 1`, so a stale or hostile hint costs a round trip and +//! changes nothing else. + +pub mod blkptr; +pub mod blockstore; +pub mod cache; +pub mod config; +pub mod dir; +pub mod dnode; +pub mod indirect; +pub mod objset; +pub mod root; +pub mod slab; +pub mod txg; + +pub use blkptr::{blkptrs_per_block, BlkPtr, Compression, Dva, BLKPTR_LEN, MAX_LEVEL}; +pub use blockstore::BlockStore; +pub use cache::{BlockCache, BlockKey, CacheStats}; +pub use config::StoreConfig; +pub use dir::{DirTxn, Dirent}; +pub use dnode::{Dnode, DnodeKind, DNODE_LEN, META_OBJID, ROOT_OBJID}; +pub use indirect::{commit_object, read_data_block, read_raw_block, resolve_ptr}; +pub use objset::ObjectSet; +pub use root::{RootRecord, RootStore, FORMAT_VERSION}; +pub use slab::{verify_and_open, FinishedSlab, SlabWriter}; +pub use txg::{Snapshot, Store, Transaction}; diff --git a/crates/s3fs-core/src/store/objset.rs b/crates/s3fs-core/src/store/objset.rs new file mode 100644 index 0000000..7c21870 --- /dev/null +++ b/crates/s3fs-core/src/store/objset.rs @@ -0,0 +1,463 @@ +//! The object set — every dnode in the filesystem, addressed by object id. +//! +//! Dnodes live in the data blocks of one special object, the **meta-dnode**. +//! Object `N` sits at slot `N % 256` of block `N / 256` (at the default record +//! size). Updating one object therefore copies one 128 KiB block and the +//! indirect path above it, and yields a new pointer at the top — and *that +//! pointer is the filesystem's Merkle root*. It is the single value the root +//! record signs, and it transitively covers every dnode, every indirect block, +//! and every byte of data. +//! +//! ```text +//! root record ──▶ meta_dnode BlkPtr ──▶ indirect ──▶ [dnode][dnode][dnode]… +//! │ +//! └──▶ that object's own tree +//! ``` +//! +//! The meta-dnode is the one object not stored in the array; its pointer lives +//! in the root record. Slot 0 is left permanently free so that slot index and +//! object id are the same number. +//! +//! ## Object ids are never reused +//! +//! Allocation is a monotonic counter. A `u64` cannot be exhausted, and not +//! reusing ids means `(objid, gen)` is a permanently stable identity — which +//! is what lets `is-same-object` and `metadata-hash` be exact, and agree +//! across mounts. Deleted objects leave a `Free` slot behind; reclaiming those +//! is a garbage-collection concern, not an allocation one. + +use std::collections::BTreeMap; + +use crate::errors::{FsError, FsResult}; + +use super::blockstore::BlockStore; +use super::dnode::{dnodes_per_block, Dnode, DnodeKind, DNODE_LEN, META_OBJID, ROOT_OBJID}; +use super::indirect::{commit_object, read_data_block}; +use super::slab::SlabWriter; + +/// Every dnode in the filesystem. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ObjectSet { + /// The dnode array's own dnode. Its `blkptr` is the Merkle root. + meta: Dnode, + /// Next object id to hand out. + next_objid: u64, +} + +impl ObjectSet { + /// A brand-new, empty object set. Object ids start after the root + /// directory's reserved id. + pub fn format(record_shift: u8) -> Self { + ObjectSet { + meta: Dnode::new(META_OBJID, DnodeKind::DnodeArray, record_shift, 0), + next_objid: ROOT_OBJID + 1, + } + } + + /// Reconstruct from the fields a root record carries. + pub fn from_root(meta: Dnode, next_objid: u64) -> FsResult { + if meta.kind != DnodeKind::DnodeArray || meta.objid != META_OBJID { + return Err(FsError::Integrity( + "objset: meta-dnode is not a dnode array", + )); + } + if next_objid <= ROOT_OBJID { + return Err(FsError::Integrity( + "objset: next_objid below the root object", + )); + } + Ok(ObjectSet { meta, next_objid }) + } + + pub fn meta(&self) -> &Dnode { + &self.meta + } + + pub fn next_objid(&self) -> u64 { + self.next_objid + } + + /// The Merkle root: the pointer covering every dnode and everything under + /// them. + pub fn merkle_root(&self) -> &super::blkptr::BlkPtr { + &self.meta.blkptr + } + + fn slots_per_block(&self) -> u64 { + dnodes_per_block(self.meta.record_size()) as u64 + } + + /// Claim a fresh object id. + pub fn alloc_objid(&mut self) -> FsResult { + let id = self.next_objid; + self.next_objid = id + .checked_add(1) + .ok_or(FsError::Invalid("object id space exhausted"))?; + Ok(id) + } + + /// Read one dnode. Unallocated slots come back as `Free`. + pub async fn get(&self, bs: &BlockStore, objid: u64) -> FsResult { + if objid == META_OBJID { + return Ok(self.meta.clone()); + } + let per = self.slots_per_block(); + let block_index = objid / per; + let slot = (objid % per) as usize; + + let block = read_data_block(bs, &self.meta, block_index).await?; + let start = slot * DNODE_LEN; + let raw = match block.get(start..start + DNODE_LEN) { + Some(raw) => raw, + // Past the end of the array: never allocated. + None => return Ok(Dnode::free(objid, self.meta.record_shift)), + }; + // An untouched slot is all zeros, which is not a decodable dnode. Treat + // it as free rather than as corruption. + if raw.iter().all(|&b| b == 0) { + return Ok(Dnode::free(objid, self.meta.record_shift)); + } + + let d = Dnode::decode(raw)?; + // The AEAD binds a block to its position, but not a dnode to its slot + // *within* a block. Without this check, two dnodes could be swapped + // inside one block and every cryptographic check would still pass. + if d.objid != objid { + return Err(FsError::Integrity("objset: dnode is in the wrong slot")); + } + Ok(d) + } + + /// Read one dnode, requiring that it is allocated. + pub async fn get_allocated(&self, bs: &BlockStore, objid: u64) -> FsResult { + let d = self.get(bs, objid).await?; + if d.kind == DnodeKind::Free { + return Err(FsError::NotFound); + } + Ok(d) + } + + /// Write `dirty` dnodes into the array and rebuild it copy-on-write. + /// + /// Blocks are staged in `writer`; the returned object set carries the new + /// Merkle root, which the caller seals into a root record. + pub async fn commit( + &self, + bs: &BlockStore, + writer: &mut SlabWriter, + dirty: BTreeMap, + ) -> FsResult { + for (objid, d) in &dirty { + if d.objid != *objid { + return Err(FsError::Invalid( + "objset: dnode objid disagrees with its key", + )); + } + if *objid == META_OBJID { + return Err(FsError::Invalid( + "objset: the meta-dnode is not an array entry", + )); + } + } + + let per = self.slots_per_block(); + let record_size = self.meta.record_size(); + let highest = dirty.keys().next_back().copied().unwrap_or(0); + let next_objid = self.next_objid.max(highest + 1); + + // Every block of the array is full-length, so slot arithmetic never + // has to reason about a short tail block. + let nblocks = next_objid.div_ceil(per); + let new_size = nblocks * record_size as u64; + + // Group by block, then read-modify-write each affected block once. + let mut by_block: BTreeMap> = BTreeMap::new(); + for (objid, d) in &dirty { + by_block.entry(objid / per).or_default().push(d); + } + + let mut dirty_blocks: BTreeMap> = BTreeMap::new(); + for (block_index, dnodes) in by_block { + let old = read_data_block(bs, &self.meta, block_index).await?; + let mut buf = old.to_vec(); + buf.resize(record_size, 0); + for d in dnodes { + let start = (d.objid % per) as usize * DNODE_LEN; + buf[start..start + DNODE_LEN].copy_from_slice(&d.encode()?); + } + dirty_blocks.insert(block_index, buf); + } + + let meta = commit_object(bs, writer, &self.meta, dirty_blocks, new_size).await?; + Ok(ObjectSet { meta, next_objid }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::backend::memory::MemoryBackend; + use crate::crypto::{KeyMaterial, MasterSecret}; + use crate::store::config::StoreConfig; + use std::sync::Arc; + + /// 4 KiB records give 8 dnodes per block, so block boundaries are easy to + /// cross in a test. + fn small_config() -> StoreConfig { + StoreConfig { + record_size: 4096, + ..Default::default() + } + } + + fn store(config: StoreConfig) -> (Arc, BlockStore) { + let backend = Arc::new(MemoryBackend::new()); + let keys = + Arc::new(KeyMaterial::derive(&MasterSecret::from_bytes([6u8; 32]), [0u8; 16]).unwrap()); + let bs = BlockStore::new(backend.clone(), keys, Arc::new(config)); + (backend, bs) + } + + async fn commit( + bs: &BlockStore, + txg: u64, + os: &ObjectSet, + dirty: BTreeMap, + ) -> ObjectSet { + let mut w = SlabWriter::new(txg, bs.config()); + let out = os.commit(bs, &mut w, dirty).await.unwrap(); + bs.write_slabs(txg, w.finish().unwrap()).await.unwrap(); + out + } + + fn file(objid: u64, size: u64) -> Dnode { + Dnode { + size, + ..Dnode::new(objid, DnodeKind::File, 12, 1000) + } + } + + #[test] + fn a_fresh_object_set_is_empty() { + let os = ObjectSet::format(12); + assert_eq!(os.next_objid(), ROOT_OBJID + 1); + assert!(os.merkle_root().is_hole()); + assert_eq!(os.meta().kind, DnodeKind::DnodeArray); + } + + #[test] + fn object_ids_are_handed_out_monotonically() { + let mut os = ObjectSet::format(12); + let a = os.alloc_objid().unwrap(); + let b = os.alloc_objid().unwrap(); + assert_eq!(b, a + 1); + assert!(a > ROOT_OBJID, "the root directory's id is reserved"); + } + + #[tokio::test] + async fn store_and_load_one_dnode() { + let (_b, bs) = store(small_config()); + let os = ObjectSet::format(12); + + let d = file(ROOT_OBJID + 1, 1234); + let os = commit(&bs, 1, &os, BTreeMap::from([(d.objid, d.clone())])).await; + + assert_eq!(os.get(&bs, d.objid).await.unwrap(), d); + assert!(!os.merkle_root().is_hole(), "the Merkle root must be live"); + } + + #[tokio::test] + async fn unallocated_slots_read_as_free() { + let (_b, bs) = store(small_config()); + let os = ObjectSet::format(12); + let os = commit(&bs, 1, &os, BTreeMap::from([(5, file(5, 10))])).await; + + // A never-written slot inside an allocated block. + assert_eq!(os.get(&bs, 4).await.unwrap().kind, DnodeKind::Free); + // A slot past the end of the array entirely. + assert_eq!(os.get(&bs, 9_999).await.unwrap().kind, DnodeKind::Free); + assert!(matches!( + os.get_allocated(&bs, 4).await, + Err(FsError::NotFound) + )); + } + + #[tokio::test] + async fn many_dnodes_across_several_blocks() { + let (_b, bs) = store(small_config()); + let os = ObjectSet::format(12); + + // 8 dnodes per 4 KiB block, so 40 objects spans several blocks. + let dirty: BTreeMap<_, _> = (2..42u64).map(|i| (i, file(i, i * 100))).collect(); + let os = commit(&bs, 1, &os, dirty).await; + + assert_eq!(os.next_objid(), 42); + for i in 2..42u64 { + let d = os.get(&bs, i).await.unwrap(); + assert_eq!(d.objid, i); + assert_eq!(d.size, i * 100, "object {i}"); + } + } + + #[tokio::test] + async fn updating_one_dnode_leaves_its_block_mates_alone() { + let (_b, bs) = store(small_config()); + let os = ObjectSet::format(12); + // Objects 2..6 share one block at 8 slots per block. + let dirty: BTreeMap<_, _> = (2..6u64).map(|i| (i, file(i, i))).collect(); + let os = commit(&bs, 1, &os, dirty).await; + + let mut changed = file(3, 999_999); + changed.mtime_nanos = 42; + let os = commit(&bs, 2, &os, BTreeMap::from([(3, changed.clone())])).await; + + assert_eq!(os.get(&bs, 3).await.unwrap(), changed); + for i in [2u64, 4, 5] { + assert_eq!( + os.get(&bs, i).await.unwrap().size, + i, + "object {i} was disturbed" + ); + } + } + + /// The Merkle root must move whenever anything beneath it does — that is + /// the entire premise of anchoring the filesystem on one hash. + #[tokio::test] + async fn the_merkle_root_changes_on_every_mutation() { + let (_b, bs) = store(small_config()); + let os = ObjectSet::format(12); + + let os1 = commit(&bs, 1, &os, BTreeMap::from([(2, file(2, 1))])).await; + let root1 = *os1.merkle_root(); + + let os2 = commit(&bs, 2, &os1, BTreeMap::from([(2, file(2, 2))])).await; + let root2 = *os2.merkle_root(); + + assert_ne!(root1.checksum, root2.checksum); + assert_ne!( + root1.dva, root2.dva, + "copy-on-write must not reuse the address" + ); + + // And the superseded object set still reads its own version. + assert_eq!(os1.get(&bs, 2).await.unwrap().size, 1); + assert_eq!(os2.get(&bs, 2).await.unwrap().size, 2); + } + + #[tokio::test] + async fn an_empty_commit_leaves_the_root_untouched() { + let (_b, bs) = store(small_config()); + let os = commit( + &bs, + 1, + &ObjectSet::format(12), + BTreeMap::from([(2, file(2, 1))]), + ) + .await; + let again = commit(&bs, 2, &os, BTreeMap::new()).await; + assert_eq!(again.merkle_root(), os.merkle_root()); + } + + #[tokio::test] + async fn committing_a_high_object_id_extends_the_array() { + let (_b, bs) = store(small_config()); + let os = commit( + &bs, + 1, + &ObjectSet::format(12), + BTreeMap::from([(1000, file(1000, 7))]), + ) + .await; + + assert_eq!(os.next_objid(), 1001); + assert_eq!(os.get(&bs, 1000).await.unwrap().size, 7); + // The intervening slots cost nothing: they are holes. + assert_eq!(os.get(&bs, 500).await.unwrap().kind, DnodeKind::Free); + } + + #[tokio::test] + async fn rejects_a_dnode_filed_under_the_wrong_key() { + let (_b, bs) = store(small_config()); + let mut w = SlabWriter::new(1, bs.config()); + let os = ObjectSet::format(12); + assert!(os + .commit(&bs, &mut w, BTreeMap::from([(5, file(6, 0))])) + .await + .is_err()); + } + + #[tokio::test] + async fn rejects_writing_the_meta_dnode_as_an_entry() { + let (_b, bs) = store(small_config()); + let mut w = SlabWriter::new(1, bs.config()); + let os = ObjectSet::format(12); + assert!(os + .commit( + &bs, + &mut w, + BTreeMap::from([(META_OBJID, file(META_OBJID, 0))]) + ) + .await + .is_err()); + } + + /// Position-binding AAD covers a whole block, not a dnode's slot inside it. + /// Two dnodes swapped within one block would pass every cryptographic + /// check, so the identity field has to be verified explicitly. + #[tokio::test] + async fn rejects_a_dnode_moved_to_another_slot() { + let (_b, bs) = store(small_config()); + let os = ObjectSet::format(12); + + // Hand-build a block where object 2's dnode sits in object 3's slot. + let mut buf = vec![0u8; 4096]; + buf[3 * DNODE_LEN..4 * DNODE_LEN].copy_from_slice(&file(2, 1).encode().unwrap()); + + let mut w = SlabWriter::new(1, bs.config()); + let meta = commit_object(&bs, &mut w, os.meta(), BTreeMap::from([(0u64, buf)]), 4096) + .await + .unwrap(); + bs.write_slabs(1, w.finish().unwrap()).await.unwrap(); + + let forged = ObjectSet::from_root(meta, 10).unwrap(); + assert!(matches!( + forged.get(&bs, 3).await, + Err(FsError::Integrity("objset: dnode is in the wrong slot")) + )); + } + + #[test] + fn from_root_rejects_a_meta_dnode_of_the_wrong_kind() { + let bad = Dnode::new(META_OBJID, DnodeKind::File, 12, 0); + assert!(ObjectSet::from_root(bad, 10).is_err()); + + let wrong_id = Dnode::new(7, DnodeKind::DnodeArray, 12, 0); + assert!(ObjectSet::from_root(wrong_id, 10).is_err()); + + let good = Dnode::new(META_OBJID, DnodeKind::DnodeArray, 12, 0); + assert!(ObjectSet::from_root(good.clone(), 10).is_ok()); + assert!( + ObjectSet::from_root(good, ROOT_OBJID).is_err(), + "next_objid must leave room for the root directory" + ); + } + + #[tokio::test] + async fn round_trips_through_from_root() { + let (_b, bs) = store(small_config()); + let os = commit( + &bs, + 1, + &ObjectSet::format(12), + BTreeMap::from([(2, file(2, 42))]), + ) + .await; + + // What a mount does: rebuild the object set from what the root record + // carries, then read through it. + let remounted = ObjectSet::from_root(os.meta().clone(), os.next_objid()).unwrap(); + assert_eq!(remounted, os); + assert_eq!(remounted.get(&bs, 2).await.unwrap().size, 42); + } +} diff --git a/crates/s3fs-core/src/store/root.rs b/crates/s3fs-core/src/store/root.rs new file mode 100644 index 0000000..64fd529 --- /dev/null +++ b/crates/s3fs-core/src/store/root.rs @@ -0,0 +1,905 @@ +//! Root records — the signed, hash-chained anchor that makes rollback +//! detectable. +//! +//! A root record is the only thing in the store that is signed, and the only +//! thing S3 Object Lock protects. Everything else derives its trust from it: +//! the record carries the meta-dnode, whose block pointer's checksum covers +//! every dnode, every indirect block, and every byte of data beneath. +//! +//! ## What each mechanism actually buys +//! +//! | Threat | Defence | +//! |---|---| +//! | Forge a root | Ed25519 signature over the whole record, verified against the key we derived — not the key the record names | +//! | Alter data under a valid root | The signed `meta_dnode` pointer's BLAKE3 checksum | +//! | Replay an old root at its own key | `seq` is checked against the key it was read from, and against an in-session floor | +//! | Destroy a root to erase history | Object Lock COMPLIANCE: the version is undeletable by every principal, including the account root | +//! | Hide a root behind a delete marker | Tip discovery and loading read the *retained* version, so a marker changes nothing | +//! | Splice two histories together | `prev_root_hash`, which the signature covers | +//! | Two writers racing | `If-None-Match: *`, then a read-back of the retained version, so exactly one wins `seq + 1` even past a delete marker | +//! +//! ## The residual risk +//! +//! A cold mount cannot distinguish "the tip is N" from "the tip is N, and the +//! store is hiding N+1". Closing it needs an anchor outside S3 (a conditional +//! counter in another service, or a floor supplied through the enclave's KMS +//! encryption context, which [`RootStore::mount`] accepts as `min_seq`). +//! +//! What *used* to make this reachable through a legal API call has been closed. +//! Object Lock protects a version, not the name it lives under: a +//! `DeleteObject` with no version id writes a delete marker, and `HEAD` and +//! `GetObject` then report the tip missing while it sits underneath, +//! undeletable. So hiding N+1 needed no lie from S3 at all — just a delete +//! nobody was allowed to refuse. [`RootStore::exists`] and [`RootStore::load`] +//! now read the retained version, which a marker cannot conceal. Hiding a root +//! now requires S3 itself to lie, which is where this paragraph always assumed +//! the bar was. +//! +//! There was a second and worse version of this, now closed. A cold mount also +//! could not distinguish "this filesystem is new" from "everything has been +//! hidden", because `Store::open` created one when it found none — so an +//! enclave pointed at an emptied store served a fresh, correctly-signed, +//! entirely wrong filesystem, and every check above passed because they all +//! attested to the new one. Mounting and creating are now separate calls, and +//! `enclave_runtime::boot` decides which it is entitled to make by checking an +//! NSM attestation the host cannot forge. +//! +//! Within a session the gap does not exist: `expected_seq` only ever rises, so +//! a root older than one already accepted is rejected outright. + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; + +use bytes::Bytes; + +use crate::backend::{Backend, PutBlobInput}; +use crate::crypto::{sign, Hash256, KeyMaterial, ED25519_PUBLIC_KEY_LEN, ED25519_SIGNATURE_LEN}; +use crate::errors::{FsError, FsResult}; + +use super::config::StoreConfig; +use super::dnode::{Dnode, DNODE_LEN}; + +/// On-disk format version. A mount refuses anything else rather than guessing. +pub const FORMAT_VERSION: u32 = 1; + +const ROOT_MAGIC: &[u8; 8] = b"S3FSROOT"; + +// Field offsets. Derived rather than written out, because hand-computed +// offsets are exactly the sort of thing that silently reads the wrong field — +// a mis-sited `signer_pubkey` would compare the wrong 32 bytes and reject +// every valid record. `field_offsets_are_stable` pins them against the encoder. +const OFF_MAGIC: usize = 0; +const OFF_FORMAT: usize = OFF_MAGIC + 8; +const OFF_SEQ: usize = OFF_FORMAT + 4; +const OFF_PREV_HASH: usize = OFF_SEQ + 8; +const OFF_TXG: usize = OFF_PREV_HASH + 32; +const OFF_TIMESTAMP: usize = OFF_TXG + 8; +const OFF_NEXT_OBJID: usize = OFF_TIMESTAMP + 8; +const OFF_FS_UUID: usize = OFF_NEXT_OBJID + 8; +const OFF_PUBKEY: usize = OFF_FS_UUID + 16; +const OFF_META_DNODE: usize = OFF_PUBKEY + ED25519_PUBLIC_KEY_LEN; + +/// Bytes covered by the signature. +const SIGNED_LEN: usize = OFF_META_DNODE + DNODE_LEN; +/// Total encoded length. +pub const ROOT_LEN: usize = SIGNED_LEN + ED25519_SIGNATURE_LEN; + +fn u64_at(bytes: &[u8], off: usize) -> u64 { + u64::from_be_bytes(bytes[off..off + 8].try_into().expect("8 bytes")) +} + +/// A committed state of the entire filesystem. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RootRecord { + pub format_version: u32, + /// Sequence number. Also the object key this record lives at, so a replay + /// to a different key is caught by comparing the two. + pub seq: u64, + /// BLAKE3 of the previous record's full encoding. Zero at genesis. + pub prev_root_hash: Hash256, + pub txg: u64, + pub timestamp_nanos: u64, + pub next_objid: u64, + pub fs_uuid: [u8; 16], + pub signer_pubkey: [u8; ED25519_PUBLIC_KEY_LEN], + /// The dnode array. Its `blkptr` checksum is the Merkle root. + pub meta_dnode: Dnode, + pub signature: [u8; ED25519_SIGNATURE_LEN], +} + +impl RootRecord { + /// Build and sign a record. + pub fn seal( + keys: &KeyMaterial, + seq: u64, + prev: Option<&RootRecord>, + txg: u64, + timestamp_nanos: u64, + meta_dnode: Dnode, + next_objid: u64, + ) -> FsResult { + let mut record = RootRecord { + format_version: FORMAT_VERSION, + seq, + prev_root_hash: prev.map(RootRecord::hash).unwrap_or(Hash256::ZERO), + txg, + timestamp_nanos, + next_objid, + fs_uuid: *keys.fs_uuid(), + signer_pubkey: *keys.public_key(), + meta_dnode, + signature: [0u8; ED25519_SIGNATURE_LEN], + }; + record.signature = sign::sign(keys.signing_key(), &record.signed_bytes()?); + Ok(record) + } + + fn signed_bytes(&self) -> FsResult> { + let mut b = Vec::with_capacity(SIGNED_LEN); + b.extend_from_slice(ROOT_MAGIC); + b.extend_from_slice(&self.format_version.to_be_bytes()); + b.extend_from_slice(&self.seq.to_be_bytes()); + b.extend_from_slice(self.prev_root_hash.as_bytes()); + b.extend_from_slice(&self.txg.to_be_bytes()); + b.extend_from_slice(&self.timestamp_nanos.to_be_bytes()); + b.extend_from_slice(&self.next_objid.to_be_bytes()); + b.extend_from_slice(&self.fs_uuid); + b.extend_from_slice(&self.signer_pubkey); + b.extend_from_slice(&self.meta_dnode.encode()?); + debug_assert_eq!(b.len(), SIGNED_LEN); + Ok(b) + } + + pub fn encode(&self) -> FsResult> { + let mut b = self.signed_bytes()?; + b.extend_from_slice(&self.signature); + Ok(b) + } + + /// Identity of this record for chaining. Covers the signature too, so a + /// link commits to one exact set of bytes. + pub fn hash(&self) -> Hash256 { + Hash256::of(&self.encode().expect("a sealed record always encodes")) + } + + /// The Merkle root: one checksum covering the whole filesystem. + pub fn merkle_root(&self) -> &Hash256 { + &self.meta_dnode.blkptr.checksum + } + + /// Decode and fully verify. + /// + /// `expected_seq` is the key the bytes were read from. Checking it against + /// the record's own `seq` is what stops a valid old record being replayed + /// at a newer key — the signature alone would happily accept that. + pub fn decode_and_verify( + bytes: &[u8], + keys: &KeyMaterial, + expected_seq: u64, + ) -> FsResult { + if bytes.len() != ROOT_LEN { + return Err(FsError::Integrity("root: wrong length")); + } + if &bytes[OFF_MAGIC..OFF_MAGIC + 8] != ROOT_MAGIC { + return Err(FsError::Integrity("root: bad magic")); + } + let format_version = u32::from_be_bytes( + bytes[OFF_FORMAT..OFF_FORMAT + 4] + .try_into() + .expect("4 bytes"), + ); + if format_version != FORMAT_VERSION { + return Err(FsError::Integrity("root: unsupported format version")); + } + + let mut signer_pubkey = [0u8; ED25519_PUBLIC_KEY_LEN]; + signer_pubkey.copy_from_slice(&bytes[OFF_PUBKEY..OFF_PUBKEY + ED25519_PUBLIC_KEY_LEN]); + // Verify against the key *we* derived, never the one the record names. + // Otherwise an attacker signs a forged record with their own key and + // ships the matching public key alongside it. + if &signer_pubkey != keys.public_key() { + return Err(FsError::Integrity("root: signed by an unexpected key")); + } + let mut signature = [0u8; ED25519_SIGNATURE_LEN]; + signature.copy_from_slice(&bytes[SIGNED_LEN..]); + sign::verify(keys.public_key(), &bytes[..SIGNED_LEN], &signature)?; + + let mut fs_uuid = [0u8; 16]; + fs_uuid.copy_from_slice(&bytes[OFF_FS_UUID..OFF_FS_UUID + 16]); + if &fs_uuid != keys.fs_uuid() { + return Err(FsError::Integrity("root: belongs to another filesystem")); + } + + let seq = u64_at(bytes, OFF_SEQ); + if seq != expected_seq { + return Err(FsError::Integrity("root: sequence does not match its key")); + } + + let record = RootRecord { + format_version, + seq, + prev_root_hash: Hash256::from_bytes( + bytes[OFF_PREV_HASH..OFF_PREV_HASH + 32] + .try_into() + .expect("32 bytes"), + ), + txg: u64_at(bytes, OFF_TXG), + timestamp_nanos: u64_at(bytes, OFF_TIMESTAMP), + next_objid: u64_at(bytes, OFF_NEXT_OBJID), + fs_uuid, + signer_pubkey, + meta_dnode: Dnode::decode(&bytes[OFF_META_DNODE..SIGNED_LEN])?, + signature, + }; + if seq == 0 && !record.prev_root_hash.is_zero() { + return Err(FsError::Integrity("root: genesis record has a predecessor")); + } + Ok(record) + } +} + +/// Reads and publishes root records against the locked bucket. +#[derive(Debug)] +pub struct RootStore { + backend: Arc, + keys: Arc, + config: Arc, + /// Floor for this session. Only ever rises, so a root older than one we + /// have already accepted can never be served to us again. + expected_seq: AtomicU64, +} + +impl RootStore { + pub fn new( + backend: Arc, + keys: Arc, + config: Arc, + ) -> Self { + RootStore { + backend, + keys, + config, + expected_seq: AtomicU64::new(0), + } + } + + /// Highest sequence accepted in this session. + pub fn expected_seq(&self) -> u64 { + self.expected_seq.load(Ordering::SeqCst) + } + + /// Is there a record at this sequence? + /// + /// Asks for the *retained* version rather than `HEAD`-ing the current one. + /// A `DeleteObject` without a version id writes a delete marker that hides + /// an Object-Locked record from `HEAD` and `GetObject` while leaving it + /// undeletable underneath — so a `HEAD` here would let a host roll the + /// chain back by hiding its tip, using nothing but a legal S3 call. + /// + /// Costs a `ListObjectVersions` per probe, on a binary search that runs + /// once per mount. Not the commit path. + /// + /// If the enclave's role lacks `s3:ListBucketVersions` this errors rather + /// than answering `false`, so a missing permission is a refusal to mount + /// and not a silent rollback. + async fn exists(&self, seq: u64) -> FsResult { + match self + .backend + .get_retained_blob(&self.config.root_key(seq)) + .await + { + Ok(_) => Ok(true), + Err(FsError::NotFound) => Ok(false), + Err(e) => Err(e), + } + } + + /// Fetch and verify the record at `seq`. + /// + /// The retained version, for the reason [`RootStore::exists`] gives: a + /// record that has been hidden behind a delete marker is still the record. + pub async fn load(&self, seq: u64) -> FsResult { + let got = self + .backend + .get_retained_blob(&self.config.root_key(seq)) + .await?; + let record = RootRecord::decode_and_verify(&got.body, &self.keys, seq)?; + let floor = self.expected_seq(); + if record.seq < floor { + return Err(FsError::Rollback { + expected: floor, + found: record.seq, + }); + } + Ok(record) + } + + /// Fetch and verify a *historical* root, bypassing the session floor. + /// + /// Only for read-only access to a past state. The floor exists to stop the + /// store rewinding the filesystem under us, so nothing that establishes + /// the live state may use this — and nothing here raises the floor either, + /// so reading an old snapshot cannot be turned into accepting one. + /// + /// Every other check still applies: signature, key, filesystem id, and + /// that the record's own sequence matches the key it was read from. + pub async fn load_snapshot(&self, seq: u64) -> FsResult { + self.verify_at(seq).await + } + + /// Locate the newest root. + /// + /// Roots are contiguous — every commit takes `seq + 1` and none can be + /// deleted — so existence is monotone in `seq` and a galloping search + /// finds the tip in O(log n) HEADs. A full LIST is not an option: after + /// years of commits there may be millions of keys, and none of them ever + /// go away. + /// + /// The `roots/latest` hint only chooses where the search starts, and it is + /// verified before being trusted, so a stale or hostile hint costs a round + /// trip and nothing else. + pub async fn find_tip(&self) -> FsResult> { + if !self.exists(0).await? { + return Ok(None); // not formatted + } + + let mut lo = 0u64; + if let Some(hint) = self.read_hint().await { + if hint > 0 && self.verify_at(hint).await.is_ok() { + lo = hint; + } + } + + let mut step = 1u64; + loop { + match lo.checked_add(step) { + Some(next) if self.exists(next).await? => { + lo = next; + step = step.saturating_mul(2); + } + _ => break, + } + } + + let mut hi = lo.saturating_add(step); // known absent + while hi - lo > 1 { + let mid = lo + (hi - lo) / 2; + if self.exists(mid).await? { + lo = mid; + } else { + hi = mid; + } + } + Ok(Some(lo)) + } + + async fn verify_at(&self, seq: u64) -> FsResult { + let got = self + .backend + .get_retained_blob(&self.config.root_key(seq)) + .await?; + RootRecord::decode_and_verify(&got.body, &self.keys, seq) + } + + async fn read_hint(&self) -> Option { + let got = self + .backend + .get_blob(&self.config.root_hint_key(), None) + .await + .ok()?; + std::str::from_utf8(&got.body).ok()?.trim().parse().ok() + } + + /// Open the filesystem at its newest root. + /// + /// `min_seq` is an externally supplied floor — from a KMS encryption + /// context, a configuration flag, or anywhere else outside the store's + /// control. It is the only thing that can close the cold-mount gap + /// described in this module's documentation. + /// + /// Returns `None` if the bucket holds no filesystem at all. + pub async fn mount(&self, min_seq: Option) -> FsResult> { + let Some(tip) = self.find_tip().await? else { + return Ok(None); + }; + if let Some(min) = min_seq { + if tip < min { + return Err(FsError::Rollback { + expected: min, + found: tip, + }); + } + } + let record = self.verify_at(tip).await?; + self.verify_chain(&record, self.config.root_chain_verify_depth) + .await?; + self.raise_floor(record.seq); + Ok(Some(record)) + } + + /// Walk `depth` links back, checking each `prev_root_hash`. + /// + /// Depth 1 catches a spliced history at the tip, which is the live attack. + /// Walking the whole chain costs one GET per root and is an audit + /// operation, not something a mount should pay for. + pub async fn verify_chain(&self, tip: &RootRecord, depth: u32) -> FsResult<()> { + let mut current = tip.clone(); + for _ in 0..depth { + if current.seq == 0 { + return Ok(()); // reached genesis + } + let prev = self.verify_at(current.seq - 1).await?; + if prev.hash() != current.prev_root_hash { + return Err(FsError::Integrity("root: broken hash chain")); + } + current = prev; + } + Ok(()) + } + + /// Publish a record, winning or losing the race for its sequence number. + /// + /// `If-None-Match: *` alone does not decide that race. It tests the + /// *current* version, and a delete marker is a current version that is not + /// an object — so a host that hides the winner's record lets a second + /// conditional create through, and two writers would each believe they + /// own the sequence. What decides it is the *retained* version: the first + /// ever written, and the one every reader loads. Versions only stack on + /// top of it, so once ours exists, whether it is that one can no longer + /// change — reading it back after the PUT is exact, where a check before + /// the PUT would race another marker. + /// + /// Losing is [`FsError::Conflict`], and the caller must treat that as + /// fatal for the mount rather than retrying: a retry would reuse the + /// transaction group number, and with it every AEAD nonce in the commit. + pub async fn publish(&self, record: &RootRecord) -> FsResult<()> { + let key = self.config.root_key(record.seq); + let body = Bytes::from(record.encode()?); + let mut input = PutBlobInput::new(key.clone(), body.clone()); + input.object_lock = self.root_retention(); + + match self.backend.put_blob_if_not_exists(input).await { + Ok(_) => {} + Err(FsError::AlreadyExists) => return Err(FsError::Conflict), + Err(e) => return Err(e), + } + // ponytail: a LIST and a GET per publish. Comparing the PUT's version + // id with the oldest listed one drops the GET, if commits need it. + if self.backend.get_retained_blob(&key).await?.body != body { + return Err(FsError::Conflict); + } + self.raise_floor(record.seq); + self.write_hint(record.seq).await; + Ok(()) + } + + fn root_retention(&self) -> Option { + self.config + .root_retention + .map(|d| crate::backend::ObjectLock { + mode: crate::backend::ObjectLockMode::Compliance, + retain_until: std::time::SystemTime::now() + d, + }) + } + + /// Best-effort tip hint. Never trusted on read, so a failure here costs a + /// slower mount and nothing more. + async fn write_hint(&self, seq: u64) { + let _ = self + .backend + .put_blob(PutBlobInput::new( + self.config.root_hint_key(), + Bytes::from(seq.to_string()), + )) + .await; + } + + fn raise_floor(&self, seq: u64) { + self.expected_seq.fetch_max(seq, Ordering::SeqCst); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::backend::memory::MemoryBackend; + use crate::crypto::MasterSecret; + use crate::store::dnode::{DnodeKind, META_OBJID}; + + fn keys_for(master: u8) -> Arc { + Arc::new(KeyMaterial::derive(&MasterSecret::from_bytes([master; 32]), [1u8; 16]).unwrap()) + } + + fn setup() -> (Arc, Arc, RootStore) { + let backend = Arc::new(MemoryBackend::new()); + let keys = keys_for(11); + let config = Arc::new(StoreConfig { + // Short retention: the tests need to be able to demonstrate that a + // committed root is refused, not to keep one for a decade. + root_retention: Some(std::time::Duration::from_secs(3600)), + ..Default::default() + }); + let rs = RootStore::new(backend.clone(), keys.clone(), config); + (backend, keys, rs) + } + + fn meta() -> Dnode { + Dnode::new(META_OBJID, DnodeKind::DnodeArray, 17, 0) + } + + fn seal(keys: &KeyMaterial, seq: u64, prev: Option<&RootRecord>) -> RootRecord { + RootRecord::seal(keys, seq, prev, seq + 1, 1_000 + seq, meta(), 2).unwrap() + } + + /// Publish a chain of `n` roots, seq 0..n-1. + async fn publish_chain(rs: &RootStore, keys: &KeyMaterial, n: u64) -> Vec { + let mut out: Vec = Vec::new(); + for seq in 0..n { + let r = seal(keys, seq, out.last()); + rs.publish(&r).await.unwrap(); + out.push(r); + } + out + } + + // ---- encoding ---------------------------------------------------------- + + #[test] + fn round_trips() { + let keys = keys_for(11); + let r = seal(&keys, 5, None); + assert_eq!(r.encode().unwrap().len(), ROOT_LEN); + assert_eq!( + RootRecord::decode_and_verify(&r.encode().unwrap(), &keys, 5).unwrap(), + r + ); + } + + /// The decoder reads by offset; the encoder writes in order. Pin every + /// boundary so a field added or resized cannot silently shift the ones + /// after it into reading the wrong bytes. + #[test] + fn field_offsets_are_stable() { + let keys = keys_for(11); + let r = RootRecord::seal(&keys, 0x1122, None, 0x3344, 0x5566, meta(), 0x7788).unwrap(); + let b = r.encode().unwrap(); + + assert_eq!(&b[OFF_MAGIC..OFF_MAGIC + 8], ROOT_MAGIC); + assert_eq!( + &b[OFF_FORMAT..OFF_FORMAT + 4], + &FORMAT_VERSION.to_be_bytes() + ); + assert_eq!(u64_at(&b, OFF_SEQ), 0x1122); + assert_eq!( + &b[OFF_PREV_HASH..OFF_PREV_HASH + 32], + Hash256::ZERO.as_bytes() + ); + assert_eq!(u64_at(&b, OFF_TXG), 0x3344); + assert_eq!(u64_at(&b, OFF_TIMESTAMP), 0x5566); + assert_eq!(u64_at(&b, OFF_NEXT_OBJID), 0x7788); + assert_eq!(&b[OFF_FS_UUID..OFF_FS_UUID + 16], keys.fs_uuid()); + assert_eq!(&b[OFF_PUBKEY..OFF_PUBKEY + 32], keys.public_key()); + assert_eq!(&b[OFF_META_DNODE..SIGNED_LEN], &meta().encode().unwrap()); + assert_eq!(&b[SIGNED_LEN..], &r.signature); + assert_eq!(b.len(), ROOT_LEN); + } + + #[test] + fn every_field_survives_the_round_trip() { + let keys = keys_for(11); + let prev = seal(&keys, 0, None); + let r = seal(&keys, 1, Some(&prev)); + let decoded = RootRecord::decode_and_verify(&r.encode().unwrap(), &keys, 1).unwrap(); + + assert_eq!(decoded.seq, 1); + assert_eq!(decoded.txg, 2); + assert_eq!(decoded.timestamp_nanos, 1001); + assert_eq!(decoded.next_objid, 2); + assert_eq!(decoded.prev_root_hash, prev.hash()); + assert_eq!(&decoded.fs_uuid, keys.fs_uuid()); + assert_eq!(&decoded.signer_pubkey, keys.public_key()); + assert_eq!(decoded.meta_dnode, meta()); + } + + // ---- verification failures --------------------------------------------- + + #[test] + fn a_tampered_record_fails_its_signature() { + let keys = keys_for(11); + let r = seal(&keys, 3, None); + let mut bytes = r.encode().unwrap(); + // Move the Merkle root: the single most valuable field to an attacker. + bytes[SIGNED_LEN - 40] ^= 0x01; + assert!(matches!( + RootRecord::decode_and_verify(&bytes, &keys, 3), + Err(FsError::Integrity(_)) + )); + } + + /// The forged-record case: an attacker signs their own record and ships + /// the matching public key. Verifying against the key in the record would + /// accept it; verifying against the key we derived does not. + #[test] + fn a_record_signed_by_another_key_is_refused() { + let ours = keys_for(11); + let theirs = keys_for(22); + let forged = seal(&theirs, 3, None); + assert!(matches!( + RootRecord::decode_and_verify(&forged.encode().unwrap(), &ours, 3), + Err(FsError::Integrity("root: signed by an unexpected key")) + )); + } + + /// A perfectly valid record replayed to a different key. + #[test] + fn a_record_replayed_to_another_key_is_refused() { + let keys = keys_for(11); + let r = seal(&keys, 3, None); + assert!(matches!( + RootRecord::decode_and_verify(&r.encode().unwrap(), &keys, 7), + Err(FsError::Integrity("root: sequence does not match its key")) + )); + } + + #[test] + fn malformed_records_are_refused() { + let keys = keys_for(11); + let r = seal(&keys, 0, None); + + assert!(RootRecord::decode_and_verify(&[], &keys, 0).is_err()); + assert!(RootRecord::decode_and_verify(&[0u8; ROOT_LEN], &keys, 0).is_err()); + + let mut short = r.encode().unwrap(); + short.pop(); + assert!(RootRecord::decode_and_verify(&short, &keys, 0).is_err()); + + let mut bad_magic = r.encode().unwrap(); + bad_magic[0] ^= 0xff; + assert!(matches!( + RootRecord::decode_and_verify(&bad_magic, &keys, 0), + Err(FsError::Integrity("root: bad magic")) + )); + + let mut bad_version = r.encode().unwrap(); + bad_version[11] = 99; + assert!(matches!( + RootRecord::decode_and_verify(&bad_version, &keys, 0), + Err(FsError::Integrity("root: unsupported format version")) + )); + } + + #[test] + fn a_record_from_another_filesystem_is_refused() { + let master = MasterSecret::from_bytes([11u8; 32]); + let ours = KeyMaterial::derive(&master, [1u8; 16]).unwrap(); + let theirs = KeyMaterial::derive(&master, [2u8; 16]).unwrap(); + let r = seal(&theirs, 0, None); + // Same master secret, different filesystem: the derived signing key + // differs, so this is caught at the signature. + assert!(RootRecord::decode_and_verify(&r.encode().unwrap(), &ours, 0).is_err()); + } + + // ---- publishing -------------------------------------------------------- + + #[tokio::test] + async fn publish_then_load() { + let (_b, keys, rs) = setup(); + let r = seal(&keys, 0, None); + rs.publish(&r).await.unwrap(); + assert_eq!(rs.load(0).await.unwrap(), r); + } + + /// Exactly one writer can take a sequence number. The loser must fail, not + /// retry: a retry would reuse the transaction group number and with it + /// every AEAD nonce in the commit. + #[tokio::test] + async fn only_one_writer_can_claim_a_sequence() { + let (_b, keys, rs) = setup(); + let first = seal(&keys, 0, None); + rs.publish(&first).await.unwrap(); + + // A different record competing for the same sequence. + let second = RootRecord::seal(&keys, 0, None, 1, 9999, meta(), 2).unwrap(); + assert!(matches!(rs.publish(&second).await, Err(FsError::Conflict))); + + // The winner's record is what remains. + assert_eq!(rs.load(0).await.unwrap(), first); + } + + /// A delete marker makes a taken sequence look free — `If-None-Match: *` + /// tests the current version, and the marker is it — so the second PUT + /// goes through. Publishing must lose anyway. This is the one place every + /// root goes through, so it covers a commit as much as a claim: a commit + /// that "won" over a hidden claim would go on sealing under the number + /// that claim's mount is about to use. + #[tokio::test] + async fn a_hidden_record_still_holds_its_sequence() { + let (backend, keys, rs) = setup(); + let first = seal(&keys, 0, None); + rs.publish(&first).await.unwrap(); + backend.delete_blob(&rs.config.root_key(0)).await.unwrap(); + + let second = RootRecord::seal(&keys, 0, None, 1, 9999, meta(), 2).unwrap(); + assert!(matches!(rs.publish(&second).await, Err(FsError::Conflict))); + assert_eq!(rs.load(0).await.unwrap(), first); + } + + /// Retention makes a record impossible to *destroy*, not impossible to + /// hide: a `DeleteObject` without a version id succeeds and writes a delete + /// marker over it. What matters is that the store looks past one. + #[tokio::test] + async fn a_published_root_can_be_hidden_but_still_loads() { + let (backend, keys, rs) = setup(); + let r = seal(&keys, 0, None); + rs.publish(&r).await.unwrap(); + + let key = rs.config.root_key(0); + backend + .delete_blob(&key) + .await + .expect("S3 permits a delete marker over a retained version"); + assert!( + matches!(backend.get_blob(&key, None).await, Err(FsError::NotFound)), + "the marker hides it from an ordinary read" + ); + + assert!(matches!( + backend + .put_blob(PutBlobInput::new(key, Bytes::from_static(b"forged"))) + .await, + Err(FsError::AccessDenied) + )); + assert_eq!(rs.load(0).await.unwrap(), r, "and the store still finds it"); + } + + // ---- tip discovery ----------------------------------------------------- + + #[tokio::test] + async fn an_unformatted_bucket_has_no_tip() { + let (_b, _k, rs) = setup(); + assert_eq!(rs.find_tip().await.unwrap(), None); + assert_eq!(rs.mount(None).await.unwrap(), None); + } + + #[tokio::test] + async fn tip_is_found_at_every_chain_length() { + for n in [1u64, 2, 3, 7, 8, 9, 33, 64] { + let (_b, keys, rs) = setup(); + publish_chain(&rs, &keys, n).await; + assert_eq!(rs.find_tip().await.unwrap(), Some(n - 1), "chain of {n}"); + } + } + + #[tokio::test] + async fn tip_is_found_without_any_hint() { + let (backend, keys, rs) = setup(); + publish_chain(&rs, &keys, 20).await; + backend + .delete_blob(&rs.config.root_hint_key()) + .await + .unwrap(); + assert_eq!(rs.find_tip().await.unwrap(), Some(19)); + } + + /// The hint chooses where the search starts and nothing else. A lie can + /// slow a mount down; it cannot change which root is accepted. + #[tokio::test] + async fn a_lying_hint_does_not_change_the_tip() { + let (backend, keys, rs) = setup(); + publish_chain(&rs, &keys, 20).await; + let hint_key = rs.config.root_hint_key(); + + for lie in ["0", "5", "999999", "garbage", ""] { + backend + .put_blob(PutBlobInput::new( + hint_key.clone(), + Bytes::from(lie.to_string()), + )) + .await + .unwrap(); + assert_eq!( + rs.find_tip().await.unwrap(), + Some(19), + "hint {lie:?} changed the tip" + ); + } + } + + // ---- mounting ---------------------------------------------------------- + + #[tokio::test] + async fn mount_returns_the_newest_root() { + let (_b, keys, rs) = setup(); + let chain = publish_chain(&rs, &keys, 5).await; + let mounted = rs.mount(None).await.unwrap().unwrap(); + assert_eq!(mounted, chain[4]); + assert_eq!(rs.expected_seq(), 4); + } + + #[tokio::test] + async fn mount_verifies_the_chain() { + let (_b, keys, rs) = setup(); + publish_chain(&rs, &keys, 3).await; + + // A record whose predecessor link is wrong: individually valid, and + // correctly signed, but not a continuation of this history. + let spliced = RootRecord::seal(&keys, 3, None, 99, 0, meta(), 2).unwrap(); + rs.publish(&spliced).await.unwrap(); + + assert!(matches!( + rs.mount(None).await, + Err(FsError::Integrity("root: broken hash chain")) + )); + } + + #[tokio::test] + async fn mount_honours_an_external_floor() { + let (_b, keys, rs) = setup(); + publish_chain(&rs, &keys, 5).await; + + assert!(rs.mount(Some(4)).await.is_ok()); + assert!(matches!( + rs.mount(Some(10)).await, + Err(FsError::Rollback { + expected: 10, + found: 4 + }) + )); + } + + /// The in-session guarantee: once a sequence has been accepted, nothing + /// older is ever served again, whatever the store offers. + #[tokio::test] + async fn an_older_root_is_refused_once_a_newer_one_is_seen() { + let (_b, keys, rs) = setup(); + publish_chain(&rs, &keys, 5).await; + rs.mount(None).await.unwrap(); + assert_eq!(rs.expected_seq(), 4); + + assert!(matches!( + rs.load(2).await, + Err(FsError::Rollback { + expected: 4, + found: 2 + }) + )); + assert!(rs.load(4).await.is_ok()); + } + + #[tokio::test] + async fn the_floor_never_falls() { + let (_b, keys, rs) = setup(); + publish_chain(&rs, &keys, 5).await; + rs.mount(None).await.unwrap(); + rs.raise_floor(2); + assert_eq!(rs.expected_seq(), 4, "the floor must only ever rise"); + } + + #[tokio::test] + async fn publishing_raises_the_floor() { + let (_b, keys, rs) = setup(); + let chain = publish_chain(&rs, &keys, 3).await; + assert_eq!(rs.expected_seq(), 2); + assert!( + rs.load(1).await.is_err(), + "our own history is now behind us" + ); + assert_eq!(rs.load(2).await.unwrap(), chain[2]); + } + + #[tokio::test] + async fn a_full_chain_walk_verifies_every_link() { + let (_b, keys, rs) = setup(); + let chain = publish_chain(&rs, &keys, 10).await; + rs.verify_chain(&chain[9], 9).await.unwrap(); + // Walking past genesis stops rather than failing. + rs.verify_chain(&chain[9], 100).await.unwrap(); + } + + #[tokio::test] + async fn the_merkle_root_is_the_meta_dnode_checksum() { + let keys = keys_for(11); + let r = seal(&keys, 0, None); + assert_eq!(r.merkle_root(), &r.meta_dnode.blkptr.checksum); + } +} diff --git a/crates/s3fs-core/src/store/slab.rs b/crates/s3fs-core/src/store/slab.rs new file mode 100644 index 0000000..ffbacd9 --- /dev/null +++ b/crates/s3fs-core/src/store/slab.rs @@ -0,0 +1,478 @@ +//! Slab packing — how a transaction group's blocks become S3 objects. +//! +//! Every block dirtied in a txg is sealed and appended to a slab buffer. When +//! a slab reaches `slab_max_bytes` a new one starts. At commit time each slab +//! becomes one immutable `slabs//` object. +//! +//! This is where the design earns its keep: a commit costs `ceil(dirty_bytes / +//! slab_max_bytes)` PUTs plus one for the root, whether the txg touched one +//! block or ten thousand. Addressing blocks by content hash instead would cost +//! one PUT per block, which for a SQLite page-write workload means a round +//! trip per 4 KiB. +//! +//! Slabs are write-once and are never revisited: an aborted or crashed commit +//! leaves orphaned slab objects that no root references, which are invisible +//! to readers and reclaimable by lifecycle policy. + +use bytes::Bytes; + +use crate::crypto::aead::{self, BlockAad, BlockNonce}; +use crate::crypto::{Hash256, KeyMaterial}; +use crate::errors::{FsError, FsResult}; + +use super::blkptr::{BlkPtr, Compression, Dva}; +use super::config::StoreConfig; + +/// One finished slab, ready to PUT. +#[derive(Debug, Clone)] +pub struct FinishedSlab { + /// Index within the txg. + pub index: u16, + pub body: Bytes, +} + +/// Accumulates sealed blocks for one transaction group. +/// +/// Holds no key material and performs no I/O; it is a pure byte-packer, which +/// keeps it trivially testable. +#[derive(Debug)] +pub struct SlabWriter { + txg: u64, + slab_max_bytes: usize, + /// Sealed slabs already closed out. + finished: Vec, + /// The slab currently being filled. + current: Vec, + /// Index of `current` within the txg. + current_index: u16, + /// Monotonic counter over every block written in this txg. Combined with + /// the txg it forms the AEAD nonce, so it must increment for every single + /// sealed block regardless of which slab the block lands in. + next_block_seq: u32, +} + +impl SlabWriter { + pub fn new(txg: u64, config: &StoreConfig) -> Self { + Self { + txg, + slab_max_bytes: config.slab_max_bytes, + finished: Vec::new(), + current: Vec::new(), + current_index: 0, + next_block_seq: 0, + } + } + + pub fn txg(&self) -> u64 { + self.txg + } + + /// Total bytes staged so far, across finished and in-progress slabs. + pub fn staged_bytes(&self) -> usize { + self.finished.iter().map(|s| s.body.len()).sum::() + self.current.len() + } + + /// Number of blocks sealed so far. + pub fn block_count(&self) -> u32 { + self.next_block_seq + } + + /// Seal `plaintext` and append it, returning the pointer that addresses it. + /// + /// The AAD is built here rather than taken from the caller so that + /// `birth_txg` is always this writer's txg. That closes the failure mode + /// where a block is sealed under one txg and its pointer records another, + /// which would produce a block that can never be opened. + pub fn write_block( + &mut self, + keys: &KeyMaterial, + objid: u64, + level: u8, + block_index: u64, + fill: u32, + plaintext: &[u8], + ) -> FsResult { + if plaintext.len() > super::blkptr::MAX_BLOCK_LEN as usize { + return Err(FsError::Invalid("block exceeds maximum block length")); + } + + let block_seq = self.next_block_seq; + // 2^32 blocks in one txg is ~500 TiB at the default record size; a txg + // that large means something has gone wrong upstream. Refuse rather + // than wrap, because wrapping would repeat a nonce. + self.next_block_seq = block_seq + .checked_add(1) + .ok_or(FsError::Invalid("too many blocks in one transaction group"))?; + + let nonce = BlockNonce::new(self.txg, block_seq); + let aad = BlockAad { + objid, + level, + block_index, + birth_txg: self.txg, + }; + let sealed = aead::seal(keys.block_key(), nonce, aad, plaintext)?; + + // Start a new slab if this block would overflow the current one. A + // block is never split across slabs — a pointer names one contiguous + // range, and splitting would cost a second GET on every read. + if !self.current.is_empty() && self.current.len() + sealed.len() > self.slab_max_bytes { + self.close_current()?; + } + + let offset = + u32::try_from(self.current.len()).expect("slab_max_bytes is validated to fit in u32"); + let len = u32::try_from(sealed.len()).expect("block length is bounded by MAX_BLOCK_LEN"); + let checksum = Hash256::of(&sealed); + self.current.extend_from_slice(&sealed); + + Ok(BlkPtr { + dva: Dva { + txg: self.txg, + slab: self.current_index, + offset, + len, + }, + logical_len: plaintext.len() as u32, + level, + compression: Compression::None, + fill, + birth_txg: self.txg, + nonce, + checksum, + }) + } + + fn close_current(&mut self) -> FsResult<()> { + let body = Bytes::from(std::mem::take(&mut self.current)); + self.finished.push(FinishedSlab { + index: self.current_index, + body, + }); + self.current_index = self + .current_index + .checked_add(1) + .ok_or(FsError::Invalid("too many slabs in one transaction group"))?; + Ok(()) + } + + /// Close out the in-progress slab and return everything to PUT. + /// + /// An empty txg yields no slabs — a commit that only changed metadata + /// already fitting in existing blocks still writes a root, but need not + /// write any data. + pub fn finish(mut self) -> FsResult> { + if !self.current.is_empty() { + self.close_current()?; + } + Ok(self.finished) + } +} + +/// Decrypt and verify one block read from a slab. +/// +/// Both checks run, in this order, and both must pass: +/// +/// 1. **Checksum against the parent pointer.** This is the Merkle link. It is +/// checked first because it needs no key, so a corrupt or substituted block +/// is rejected before the cipher ever sees it. +/// 2. **AEAD open with position-binding AAD.** This catches a block that is +/// genuine but served from the wrong place in the tree. +pub fn verify_and_open( + keys: &KeyMaterial, + ptr: &BlkPtr, + objid: u64, + block_index: u64, + stored: &[u8], +) -> FsResult> { + if stored.len() != ptr.dva.len as usize { + return Err(FsError::Integrity("block: short read from slab")); + } + Hash256::of(stored).verify(&ptr.checksum, "block: checksum mismatch")?; + + let aad = BlockAad { + objid, + level: ptr.level, + block_index, + birth_txg: ptr.birth_txg, + }; + let plaintext = aead::open(keys.block_key(), ptr.nonce, aad, stored)?; + if plaintext.len() != ptr.logical_len as usize { + return Err(FsError::Integrity("block: logical length mismatch")); + } + Ok(plaintext) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::crypto::MasterSecret; + + fn keys() -> KeyMaterial { + KeyMaterial::derive(&MasterSecret::from_bytes([3u8; 32]), [0u8; 16]).unwrap() + } + + fn config() -> StoreConfig { + StoreConfig::default() + } + + /// Pull a block's stored bytes back out of the slab it was packed into. + fn extract(slabs: &[FinishedSlab], ptr: &BlkPtr) -> Vec { + let slab = slabs + .iter() + .find(|s| s.index == ptr.dva.slab) + .expect("slab exists"); + let r = ptr.dva.range(); + slab.body[r.start as usize..r.end as usize].to_vec() + } + + #[test] + fn write_then_read_back_round_trips() { + let k = keys(); + let mut w = SlabWriter::new(7, &config()); + let data = vec![0xabu8; 4096]; + let ptr = w.write_block(&k, 10, 0, 0, 1, &data).unwrap(); + let slabs = w.finish().unwrap(); + + assert_eq!(slabs.len(), 1); + assert_eq!(ptr.dva.txg, 7); + assert_eq!(ptr.birth_txg, 7); + assert_eq!(ptr.logical_len, 4096); + assert_eq!(ptr.dva.len, 4096 + crate::crypto::TAG_LEN as u32); + + let stored = extract(&slabs, &ptr); + assert_eq!(verify_and_open(&k, &ptr, 10, 0, &stored).unwrap(), data); + } + + #[test] + fn blocks_are_packed_contiguously() { + let k = keys(); + let mut w = SlabWriter::new(1, &config()); + let a = w.write_block(&k, 1, 0, 0, 1, &[1u8; 100]).unwrap(); + let b = w.write_block(&k, 1, 0, 1, 1, &[2u8; 200]).unwrap(); + + assert_eq!(a.dva.offset, 0); + assert_eq!(b.dva.offset, a.dva.len, "second block follows the first"); + assert_eq!(a.dva.slab, b.dva.slab); + + let slabs = w.finish().unwrap(); + assert_eq!(slabs[0].body.len() as u32, a.dva.len + b.dva.len); + } + + #[test] + fn empty_txg_produces_no_slabs() { + let w = SlabWriter::new(1, &config()); + assert!(w.finish().unwrap().is_empty()); + } + + #[test] + fn oversized_txg_rolls_to_a_new_slab() { + let k = keys(); + // A stored block is 4096 + 16 bytes of AEAD tag = 4112, so a cap of + // 8300 admits exactly two per slab and the third must roll over. + const STORED: u32 = 4096 + crate::crypto::TAG_LEN as u32; + let cfg = StoreConfig { + slab_max_bytes: 2 * STORED as usize + 76, + record_size: 4096, + ..Default::default() + }; + let mut w = SlabWriter::new(1, &cfg); + + let ptrs: Vec<_> = (0..3) + .map(|i| w.write_block(&k, 1, 0, i, 1, &[i as u8; 4096]).unwrap()) + .collect(); + + assert_eq!((ptrs[0].dva.slab, ptrs[0].dva.offset), (0, 0)); + assert_eq!((ptrs[1].dva.slab, ptrs[1].dva.offset), (0, STORED)); + assert_eq!( + (ptrs[2].dva.slab, ptrs[2].dva.offset), + (1, 0), + "a new slab restarts the offset" + ); + + let slabs = w.finish().unwrap(); + assert_eq!(slabs.len(), 2); + for (i, p) in ptrs.iter().enumerate() { + let stored = extract(&slabs, p); + assert_eq!( + verify_and_open(&k, p, 1, i as u64, &stored).unwrap(), + vec![i as u8; 4096], + "block {i} failed to verify" + ); + } + } + + /// A block never straddles two slabs, because a pointer names exactly one + /// contiguous range and a split would double the reads. + #[test] + fn a_block_is_never_split_across_slabs() { + let k = keys(); + let cfg = StoreConfig { + slab_max_bytes: 5000, + record_size: 4096, + ..Default::default() + }; + let mut w = SlabWriter::new(1, &cfg); + let a = w.write_block(&k, 1, 0, 0, 1, &[0u8; 4096]).unwrap(); + let b = w.write_block(&k, 1, 0, 1, 1, &[0u8; 4096]).unwrap(); + assert_ne!(a.dva.slab, b.dva.slab); + + let slabs = w.finish().unwrap(); + for (s, p) in slabs.iter().zip([a, b]) { + assert_eq!(s.body.len(), p.dva.len as usize); + } + } + + /// The nonce-uniqueness invariant, checked at the level where it is + /// actually established: the sequence counter spans the whole txg, so + /// rolling to a new slab must not restart it. + #[test] + fn block_sequence_is_unique_across_the_whole_txg() { + let k = keys(); + let cfg = StoreConfig { + slab_max_bytes: 8192, + record_size: 4096, + ..Default::default() + }; + let mut w = SlabWriter::new(1, &cfg); + let mut nonces = std::collections::HashSet::new(); + for i in 0..10u64 { + let p = w.write_block(&k, 1, 0, i, 1, &[0u8; 4096]).unwrap(); + assert!( + nonces.insert(*p.nonce.as_bytes()), + "nonce reused at block {i}" + ); + } + assert_eq!(w.block_count(), 10); + } + + #[test] + fn identical_plaintext_in_one_txg_gets_distinct_ciphertext() { + let k = keys(); + let mut w = SlabWriter::new(1, &config()); + let a = w.write_block(&k, 1, 0, 0, 1, &[7u8; 1024]).unwrap(); + let b = w.write_block(&k, 1, 0, 1, 1, &[7u8; 1024]).unwrap(); + assert_ne!( + a.checksum, b.checksum, + "distinct nonces must yield distinct ciphertext" + ); + } + + #[test] + fn staged_bytes_tracks_both_finished_and_current() { + let k = keys(); + let cfg = StoreConfig { + slab_max_bytes: 8192, + record_size: 4096, + ..Default::default() + }; + let mut w = SlabWriter::new(1, &cfg); + assert_eq!(w.staged_bytes(), 0); + w.write_block(&k, 1, 0, 0, 1, &[0u8; 4096]).unwrap(); + assert_eq!(w.staged_bytes(), 4096 + 16); + w.write_block(&k, 1, 0, 1, 1, &[0u8; 4096]).unwrap(); + w.write_block(&k, 1, 0, 2, 1, &[0u8; 4096]).unwrap(); + assert_eq!(w.staged_bytes(), 3 * (4096 + 16)); + } + + // ---- verification failures --------------------------------------------- + + #[test] + fn rejects_corrupted_bytes() { + let k = keys(); + let mut w = SlabWriter::new(1, &config()); + let ptr = w.write_block(&k, 1, 0, 0, 1, b"payload").unwrap(); + let slabs = w.finish().unwrap(); + + let mut stored = extract(&slabs, &ptr); + stored[0] ^= 0x01; + assert!(matches!( + verify_and_open(&k, &ptr, 1, 0, &stored), + Err(FsError::Integrity("block: checksum mismatch")) + )); + } + + #[test] + fn rejects_short_read() { + let k = keys(); + let mut w = SlabWriter::new(1, &config()); + let ptr = w.write_block(&k, 1, 0, 0, 1, b"payload").unwrap(); + let slabs = w.finish().unwrap(); + + let stored = extract(&slabs, &ptr); + assert!(matches!( + verify_and_open(&k, &ptr, 1, 0, &stored[..stored.len() - 1]), + Err(FsError::Integrity("block: short read from slab")) + )); + } + + /// The substitution attack: swap two genuine, correctly-checksummed blocks + /// and adjust the pointer to match. The checksum passes; the AAD does not. + #[test] + fn rejects_a_genuine_block_moved_to_another_position() { + let k = keys(); + let mut w = SlabWriter::new(1, &config()); + let a = w.write_block(&k, 1, 0, 0, 1, b"block a").unwrap(); + let b = w.write_block(&k, 7, 0, 3, 1, b"block b").unwrap(); + let slabs = w.finish().unwrap(); + + // Serve block b's bytes under a pointer that has b's checksum but + // claims a's position. The Merkle check passes. + let stored_b = extract(&slabs, &b); + assert!(Hash256::of(&stored_b).verify(&b.checksum, "x").is_ok()); + + // Wrong objid. + assert!(verify_and_open(&k, &b, 1, 3, &stored_b).is_err()); + // Wrong block index. + assert!(verify_and_open(&k, &b, 7, 0, &stored_b).is_err()); + // Right position still works. + assert!(verify_and_open(&k, &b, 7, 3, &stored_b).is_ok()); + assert!(verify_and_open(&k, &a, 1, 0, &extract(&slabs, &a)).is_ok()); + } + + #[test] + fn rejects_wrong_key() { + let mut w = SlabWriter::new(1, &config()); + let ptr = w.write_block(&keys(), 1, 0, 0, 1, b"payload").unwrap(); + let slabs = w.finish().unwrap(); + + let other = KeyMaterial::derive(&MasterSecret::from_bytes([9u8; 32]), [0u8; 16]).unwrap(); + assert!(verify_and_open(&other, &ptr, 1, 0, &extract(&slabs, &ptr)).is_err()); + } + + #[test] + fn rejects_tampered_logical_length() { + let k = keys(); + let mut w = SlabWriter::new(1, &config()); + let mut ptr = w.write_block(&k, 1, 0, 0, 1, b"payload").unwrap(); + let slabs = w.finish().unwrap(); + let stored = extract(&slabs, &ptr); + + ptr.logical_len -= 1; + assert!(matches!( + verify_and_open(&k, &ptr, 1, 0, &stored), + Err(FsError::Integrity("block: logical length mismatch")) + )); + } + + #[test] + fn zero_length_block_round_trips() { + let k = keys(); + let mut w = SlabWriter::new(1, &config()); + let ptr = w.write_block(&k, 1, 0, 0, 0, b"").unwrap(); + let slabs = w.finish().unwrap(); + assert_eq!(ptr.logical_len, 0); + assert!(verify_and_open(&k, &ptr, 1, 0, &extract(&slabs, &ptr)) + .unwrap() + .is_empty()); + } + + #[test] + fn pointers_decode_after_a_round_trip_through_bytes() { + let k = keys(); + let mut w = SlabWriter::new(42, &config()); + let ptr = w.write_block(&k, 5, 2, 9, 3, &[1u8; 1000]).unwrap(); + assert_eq!(BlkPtr::decode(&ptr.encode()).unwrap(), ptr); + assert!(!ptr.is_hole()); + } +} diff --git a/crates/s3fs-core/src/store/txg.rs b/crates/s3fs-core/src/store/txg.rs new file mode 100644 index 0000000..a2479e2 --- /dev/null +++ b/crates/s3fs-core/src/store/txg.rs @@ -0,0 +1,1151 @@ +//! Transaction groups — the commit protocol, and the mounted filesystem. +//! +//! A commit is the only way state changes, and it is atomic at exactly one +//! point: the conditional PUT of the root record. +//! +//! ```text +//! 0. first commit of a session only: claim a transaction group (below) +//! 1. consume a transaction group number (never reused, see below) +//! 2. rebuild the dnode array copy-on-write (staged in memory) +//! 3. PUT every slab, and wait for all of them ← durability barrier +//! 4. seal and sign a root record +//! 5. PUT roots/ with If-None-Match: * ← the atomic commit point +//! ``` +//! +//! **Crash consistency** falls out of the ordering. Slabs written without a +//! root are orphans: nothing references them, no reader can reach them, and +//! they are reclaimable by lifecycle policy. A root that exists implies every +//! block it names was already durable, because step 3 completed before step 4 +//! began. There is no journal and no fsck. +//! +//! **Losing step 5 is fatal, not retryable.** Another writer took the sequence +//! number. Retrying would reuse this transaction group, and with it every AEAD +//! nonce in the commit — which for AES-GCM leaks the authentication subkey. +//! So the mount is poisoned and every later operation fails. +//! +//! ## Transaction group numbers are never reused +//! +//! This is the invariant the whole encryption scheme rests on: a block nonce +//! is `(txg, seq)` under one key, so sealing two different blocks under the +//! same transaction group is the GCM nonce-reuse break. Two things threaten +//! it, and both come from a number being burned without a root recording it: +//! a crash *between* steps 3 and 4, and a second mount opened at the same tip. +//! A naive remount, or the second mount, would resume at `tip + 1` — the very +//! number whose nonces the orphaned slabs already spent. +//! +//! The rule that closes both: **every burned transaction group is at most one +//! past the highest published root.** A session earns the right to seal +//! blocks by first publishing a *claim* — a root that names `tip + 1` and no +//! new blocks — and after any transaction that did not end in a published +//! root, whether it failed or was simply dropped, it must claim again before +//! sealing under the next number. The claim goes through the same +//! [`RootStore::publish`] as a commit, so of two mounts at one tip exactly one +//! claims — a delete marker over the winner's claim included — and the other +//! is poisoned before it has encrypted a byte. And because the rule is +//! enforced in the locked root chain, nothing the host can delete from the +//! data bucket weakens it. + +use std::collections::BTreeMap; +use std::sync::Arc; +use std::time::{SystemTime, UNIX_EPOCH}; + +use tokio::sync::Mutex; + +use crate::backend::Backend; +use crate::crypto::KeyMaterial; +use crate::errors::{FsError, FsResult}; + +use super::blockstore::BlockStore; +use super::config::StoreConfig; +use super::dnode::{Dnode, DnodeKind, ROOT_OBJID}; +use super::objset::ObjectSet; +use super::root::{RootRecord, RootStore}; +use super::slab::SlabWriter; + +/// Default permissions for the root directory. +const ROOT_DIR_MODE: u32 = 0o040755; + +fn now_nanos() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|d| d.as_nanos() as u64) + .unwrap_or(0) +} + +#[derive(Debug)] +struct State { + objset: ObjectSet, + root: RootRecord, + next_txg: u64, + /// Whether the root chain already records `next_txg - 1`, so sealing + /// under `next_txg` cannot collide with anything an earlier session or a + /// rival mount burned. True only when the last number handed out ended + /// in a published root: false at mount, and from `begin` until `commit`. + claimed: bool, + /// Once set, every operation fails with this. A poisoned mount has either + /// lost a race for its sequence number or seen the store misbehave; in + /// both cases continuing would risk silently diverging from what is + /// actually committed. + poison: Option, +} + +/// A mounted filesystem: the block store, the root chain, and the currently +/// committed object set. +#[derive(Debug)] +pub struct Store { + blocks: Arc, + roots: Arc, + keys: Arc, + state: Mutex, +} + +impl Store { + /// Mount an existing filesystem. Fails with [`FsError::NoFilesystem`] if + /// the store is empty. + /// + /// It used to format in that case, and that was the single most dangerous + /// line in the crate: an enclave pointed at a store that answered + /// "nothing" created a fresh filesystem and served it. Every check below + /// then passed — the signature, the hash chain, the Merkle root — because + /// they all attested to the *new* filesystem, while the guest saw an empty + /// database where its data should have been. + /// + /// Creating a filesystem is now something a caller has to ask for by name, + /// because only the caller can know whether it is entitled to. Inside an + /// enclave that entitlement is a state-origin receipt. + pub async fn open_existing( + data: Arc, + roots: Arc, + keys: Arc, + config: Arc, + min_seq: Option, + ) -> FsResult { + config.validate()?; + let blocks = Arc::new(BlockStore::new(data, keys.clone(), config.clone())); + let root_store = Arc::new(RootStore::new(roots, keys.clone(), config.clone())); + + match root_store.mount(min_seq).await? { + Some(root) => Store::from_root(blocks, root_store, keys, root).await, + None => Err(FsError::NoFilesystem), + } + } + + /// Create a brand-new filesystem in an empty store. + /// + /// Refuses if one already exists, so a caller that has misjudged which + /// mode it is in cannot overwrite a filesystem it should have mounted. + pub async fn create( + data: Arc, + roots: Arc, + keys: Arc, + config: Arc, + ) -> FsResult { + config.validate()?; + let blocks = Arc::new(BlockStore::new(data, keys.clone(), config.clone())); + let root_store = Arc::new(RootStore::new(roots, keys.clone(), config.clone())); + + if root_store.mount(None).await?.is_some() { + return Err(FsError::AlreadyExists); + } + Store::format(blocks, root_store, keys, config).await + } + + async fn from_root( + blocks: Arc, + roots: Arc, + keys: Arc, + root: RootRecord, + ) -> FsResult { + let objset = ObjectSet::from_root(root.meta_dnode.clone(), root.next_objid)?; + let next_txg = root + .txg + .checked_add(1) + .ok_or(FsError::Invalid("transaction group space exhausted"))?; + Ok(Store { + blocks, + roots, + keys, + state: Mutex::new(State { + objset, + root, + next_txg, + claimed: false, + poison: None, + }), + }) + } + + /// Create a brand-new filesystem: an empty root directory, committed as + /// root record 0. + async fn format( + blocks: Arc, + roots: Arc, + keys: Arc, + config: Arc, + ) -> FsResult { + let shift = config.record_shift(); + let now = now_nanos(); + let txg = 1; + + let mut root_dir = Dnode::new(ROOT_OBJID, DnodeKind::Dir, shift, now); + root_dir.mode = ROOT_DIR_MODE; + + let mut writer = SlabWriter::new(txg, &config); + let objset = ObjectSet::format(shift) + .commit( + &blocks, + &mut writer, + BTreeMap::from([(ROOT_OBJID, root_dir)]), + ) + .await?; + blocks.write_slabs(txg, writer.finish()?).await?; + + let root = RootRecord::seal( + &keys, + 0, + None, + txg, + now, + objset.meta().clone(), + objset.next_objid(), + )?; + roots.publish(&root).await?; + + Ok(Store { + blocks, + roots, + keys, + state: Mutex::new(State { + objset, + root, + next_txg: txg + 1, + claimed: true, + poison: None, + }), + }) + } + + pub fn blocks(&self) -> &Arc { + &self.blocks + } + + pub fn roots(&self) -> &Arc { + &self.roots + } + + /// Snapshot of the committed object set. Reads go through this. + pub async fn objset(&self) -> FsResult { + let st = self.state.lock().await; + st.check()?; + Ok(st.objset.clone()) + } + + /// The currently committed root record. + /// Fetch a historical root without disturbing the session floor. + /// + /// The boot machine needs the seq-0 record to identify *which history* + /// this filesystem is, and that record is by definition below the floor. + pub async fn snapshot_root(&self, seq: u64) -> FsResult { + self.roots.load_snapshot(seq).await + } + + pub async fn root(&self) -> FsResult { + let st = self.state.lock().await; + st.check()?; + Ok(st.root.clone()) + } + + /// Claim an object id for a new file, directory, or symlink. + pub async fn reserve_objid(&self) -> FsResult { + let mut st = self.state.lock().await; + st.check()?; + st.objset.alloc_objid() + } + + /// Whether this mount has been poisoned, and why. + pub async fn poison(&self) -> Option { + self.state.lock().await.poison.clone() + } + + /// Open a past state for reading. + /// + /// Every root record ever committed is a complete, self-verifying snapshot + /// — copy-on-write means the blocks it names were never overwritten. So a + /// snapshot costs nothing to keep and nothing to take; it is simply an + /// older root that was never deleted. + /// + /// Read-only by construction: a [`Snapshot`] has no transaction, and + /// opening one does not disturb the live mount or its rollback floor. + pub async fn snapshot(&self, seq: u64) -> FsResult { + let root = self.roots.load_snapshot(seq).await?; + let objset = ObjectSet::from_root(root.meta_dnode.clone(), root.next_objid)?; + Ok(Snapshot { root, objset }) + } + + /// The newest `limit` root records, most recent first. + pub async fn list_snapshots(&self, limit: usize) -> FsResult> { + if limit == 0 { + return Ok(Vec::new()); + } + let tip = self.root().await?.seq; + let mut out = Vec::new(); + let mut seq = tip; + loop { + out.push(self.roots.load_snapshot(seq).await?); + if seq == 0 || out.len() >= limit { + return Ok(out); + } + seq -= 1; + } + } + + /// Publish a root naming `next_txg` and no new blocks, so that number is + /// on record before anything is sealed under the one after it. + /// + /// Losing the race for the sequence number means another mount is live at + /// this tip; that is fatal for the same reason losing a commit is, except + /// that here nothing has been encrypted yet, which is the point. + async fn claim(&self, state: &mut State) -> FsResult<()> { + let txg = state.next_txg; + let root = RootRecord::seal( + &self.keys, + state.root.seq + 1, + Some(&state.root), + txg, + now_nanos(), + state.objset.meta().clone(), + state.objset.next_objid(), + )?; + if let Err(e) = self.roots.publish(&root).await { + if !e.is_transient() { + state.poison = Some(e.clone()); + } + return Err(e); + } + state.root = root; + state.next_txg = txg + .checked_add(1) + .ok_or(FsError::Invalid("transaction group space exhausted"))?; + state.claimed = true; + Ok(()) + } + + /// Begin a transaction. + /// + /// The returned handle holds the store's lock, so transactions serialise. + /// It also owns the transaction group's slab writer: object blocks and the + /// dnodes naming them **must** be staged through the same transaction, or + /// the root would reference blocks written under a different transaction + /// group — a number that may never have been committed at all. + pub async fn begin(&self) -> FsResult> { + let mut state = self.state.lock().await; + state.check()?; + if !state.claimed { + self.claim(&mut state).await?; + } + + // Consume the transaction group up front and never give it back. An + // abandoned transaction burns its number, which is exactly right: + // nothing may ever reuse it. + let txg = state.next_txg; + state.next_txg = txg + .checked_add(1) + .ok_or(FsError::Invalid("transaction group space exhausted"))?; + // Handed out, not yet on record. Stays false if this transaction is + // dropped or fails, and the next `begin` claims before sealing. + state.claimed = false; + + Ok(Transaction { + store: self, + writer: SlabWriter::new(txg, self.blocks.config()), + state, + txg, + dirty: BTreeMap::new(), + }) + } + + /// Commit a set of modified dnodes whose blocks are already durable. + /// + /// Convenience for callers that change only dnode metadata. Anything that + /// writes object blocks must use [`Store::begin`] instead. + pub async fn commit(&self, dirty: BTreeMap) -> FsResult { + if dirty.is_empty() { + return self.root().await; + } + let mut txn = self.begin().await?; + for (_, d) in dirty { + txn.stage(d); + } + txn.commit().await + } +} + +/// A past state of the filesystem, opened read-only. +#[derive(Debug, Clone)] +pub struct Snapshot { + /// The root record this snapshot was taken from. + pub root: RootRecord, + objset: ObjectSet, +} + +impl Snapshot { + pub fn objset(&self) -> &ObjectSet { + &self.objset + } +} + +/// One transaction group in progress. +/// +/// Stage object blocks through [`Transaction::writer`] and the dnodes that +/// name them through [`Transaction::stage`], then [`Transaction::commit`]. +/// Dropping without committing abandons the work: nothing was published, and +/// the staged blocks were never written. +#[derive(Debug)] +pub struct Transaction<'a> { + store: &'a Store, + state: tokio::sync::MutexGuard<'a, State>, + txg: u64, + writer: SlabWriter, + dirty: BTreeMap, +} + +impl Transaction<'_> { + pub fn txg(&self) -> u64 { + self.txg + } + + pub fn blocks(&self) -> &Arc { + &self.store.blocks + } + + /// The slab writer for this transaction group. + pub fn writer(&mut self) -> &mut SlabWriter { + &mut self.writer + } + + /// The committed object set this transaction builds on. + pub fn objset(&self) -> &ObjectSet { + &self.state.objset + } + + /// The root record this transaction will supersede. + pub fn base_root(&self) -> &RootRecord { + &self.state.root + } + + /// Claim an object id for a new object. + pub fn reserve_objid(&mut self) -> FsResult { + self.state.objset.alloc_objid() + } + + /// Record a modified dnode. + pub fn stage(&mut self, dnode: Dnode) { + self.dirty.insert(dnode.objid, dnode); + } + + /// Whether anything at all has been staged. + pub fn is_empty(&self) -> bool { + self.dirty.is_empty() && self.writer.block_count() == 0 + } + + /// Publish. See the module documentation for the ordering and why losing + /// the final conditional PUT is fatal rather than retryable. + pub async fn commit(mut self) -> FsResult { + if self.is_empty() { + return Ok(self.state.root.clone()); + } + match self.run().await { + Ok((objset, root)) => { + self.state.objset = objset; + self.state.root = root.clone(); + self.state.claimed = true; + Ok(root) + } + Err(e) => { + // A transient failure before the root PUT leaves the store + // exactly as it was — the slabs are orphans — so the mount + // stays usable and the caller may retry with a fresh + // transaction group, once the next `begin` has claimed it: + // this one was burned with no root to say so. Anything else + // means we no longer know what is committed. + if !e.is_transient() { + self.state.poison = Some(e.clone()); + } + Err(e) + } + } + } + + async fn run(&mut self) -> FsResult<(ObjectSet, RootRecord)> { + let blocks = &self.store.blocks; + let dirty = std::mem::take(&mut self.dirty); + let objset = self + .state + .objset + .commit(blocks, &mut self.writer, dirty) + .await?; + + // Durability barrier: every block the root will name must be readable + // before the root exists, or a crash here would publish a root + // pointing at data that was never written. + let slabs = std::mem::replace(&mut self.writer, SlabWriter::new(self.txg, blocks.config())) + .finish()?; + blocks.write_slabs(self.txg, slabs).await?; + + let root = RootRecord::seal( + &self.store.keys, + self.state.root.seq + 1, + Some(&self.state.root), + self.txg, + now_nanos(), + objset.meta().clone(), + objset.next_objid(), + )?; + self.store.roots.publish(&root).await?; + Ok((objset, root)) + } +} + +impl State { + fn check(&self) -> FsResult<()> { + match &self.poison { + Some(e) => Err(e.clone()), + None => Ok(()), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::backend::memory::MemoryBackend; + use crate::backend::PutBlobInput; + use crate::crypto::MasterSecret; + use crate::store::dir::{DirTxn, Dirent}; + use bytes::Bytes; + + struct Harness { + data: Arc, + roots: Arc, + keys: Arc, + config: Arc, + } + + impl Harness { + fn new() -> Self { + Harness { + data: Arc::new(MemoryBackend::new()), + roots: Arc::new(MemoryBackend::new()), + keys: Arc::new( + KeyMaterial::derive(&MasterSecret::from_bytes([13u8; 32]), [3u8; 16]).unwrap(), + ), + config: Arc::new(StoreConfig { + record_size: 4096, + root_retention: Some(std::time::Duration::from_secs(3600)), + ..Default::default() + }), + } + } + + /// Create on first use, mount thereafter — what `Store::open` used to + /// do implicitly, now spelled out because the implicit version was the + /// bug. + async fn open(&self) -> FsResult { + match self.mount(None).await { + Err(FsError::NoFilesystem) => { + Store::create( + self.data.clone(), + self.roots.clone(), + self.keys.clone(), + self.config.clone(), + ) + .await + } + other => other, + } + } + + async fn mount(&self, min_seq: Option) -> FsResult { + Store::open_existing( + self.data.clone(), + self.roots.clone(), + self.keys.clone(), + self.config.clone(), + min_seq, + ) + .await + } + } + + fn file(objid: u64, size: u64) -> Dnode { + Dnode { + size, + ..Dnode::new(objid, DnodeKind::File, 12, 0) + } + } + + #[tokio::test] + async fn formatting_creates_a_root_directory() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + + let root = store.root().await.unwrap(); + assert_eq!(root.seq, 0); + assert!(root.prev_root_hash.is_zero(), "genesis has no predecessor"); + + let objset = store.objset().await.unwrap(); + let dir = objset + .get_allocated(store.blocks(), ROOT_OBJID) + .await + .unwrap(); + assert_eq!(dir.kind, DnodeKind::Dir); + assert_eq!(dir.mode, ROOT_DIR_MODE); + } + + #[tokio::test] + async fn reopening_finds_the_same_filesystem() { + let h = Harness::new(); + let first = h.open().await.unwrap(); + let root = first.root().await.unwrap(); + drop(first); + + let second = h.open().await.unwrap(); + assert_eq!(second.root().await.unwrap(), root, "must not re-format"); + } + + #[tokio::test] + async fn each_commit_advances_the_sequence_and_the_merkle_root() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let genesis = store.root().await.unwrap(); + + let objid = store.reserve_objid().await.unwrap(); + let r1 = store + .commit(BTreeMap::from([(objid, file(objid, 100))])) + .await + .unwrap(); + assert_eq!(r1.seq, 1); + assert_eq!(r1.prev_root_hash, genesis.hash()); + assert_ne!(r1.merkle_root(), genesis.merkle_root()); + assert!(r1.txg > genesis.txg); + + let r2 = store + .commit(BTreeMap::from([(objid, file(objid, 200))])) + .await + .unwrap(); + assert_eq!(r2.seq, 2); + assert_eq!(r2.prev_root_hash, r1.hash()); + assert_ne!(r2.merkle_root(), r1.merkle_root()); + } + + #[tokio::test] + async fn committed_state_survives_a_remount() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let objid = store.reserve_objid().await.unwrap(); + store + .commit(BTreeMap::from([(objid, file(objid, 4242))])) + .await + .unwrap(); + drop(store); + + let store = h.open().await.unwrap(); + let objset = store.objset().await.unwrap(); + let d = objset.get_allocated(store.blocks(), objid).await.unwrap(); + assert_eq!(d.size, 4242); + } + + #[tokio::test] + async fn an_empty_commit_costs_nothing() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let before = store.root().await.unwrap(); + let objects_before = h.roots.object_count(); + + let after = store.commit(BTreeMap::new()).await.unwrap(); + assert_eq!(after, before); + assert_eq!(h.roots.object_count(), objects_before, "no root published"); + } + + #[tokio::test] + async fn a_directory_survives_a_remount() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + + let mut txn = store.begin().await.unwrap(); + // Own the handle rather than borrowing it from the transaction, so + // reads can interleave with staging blocks into the same transaction. + let blocks = txn.blocks().clone(); + let objid = txn.reserve_objid().unwrap(); + let dir = txn + .objset() + .get_allocated(&blocks, ROOT_OBJID) + .await + .unwrap(); + + let mut dirtxn = DirTxn::load(&blocks, &dir).await.unwrap(); + dirtxn + .insert(Dirent { + name: "hello.txt".into(), + objid, + kind: DnodeKind::File, + }) + .await + .unwrap(); + + // The directory's blocks and the dnodes naming them belong to one + // transaction group. Staging them separately would leave the root + // pointing at blocks from a group that may never be committed. + let updated_dir = dirtxn.finish(txn.writer()).await.unwrap(); + txn.stage(updated_dir); + txn.stage(file(objid, 5)); + txn.commit().await.unwrap(); + drop(store); + + let store = h.open().await.unwrap(); + let objset = store.objset().await.unwrap(); + let dir = objset + .get_allocated(store.blocks(), ROOT_OBJID) + .await + .unwrap(); + let mut txn = DirTxn::load(store.blocks(), &dir).await.unwrap(); + assert_eq!( + txn.lookup("hello.txt").await.unwrap().map(|e| e.objid), + Some(objid) + ); + } + + // ---- transaction group discipline -------------------------------------- + + /// The crash that matters: slabs written, no root published. Resuming at + /// the tip's transaction group plus one would land on the number whose + /// nonces were already spent. + #[tokio::test] + async fn a_crash_between_slabs_and_root_does_not_reuse_a_transaction_group() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let objid = store.reserve_objid().await.unwrap(); + store + .commit(BTreeMap::from([(objid, file(objid, 1))])) + .await + .unwrap(); + + // The real thing: slabs land, the root PUT dies, the process dies. + h.roots.fail_next_put(FsError::IoTimeout); + let objid = store.reserve_objid().await.unwrap(); + assert!(store + .commit(BTreeMap::from([(objid, file(objid, 2))])) + .await + .is_err()); + let orphan_txg = store.state.lock().await.next_txg - 1; + let orphan_key = h.config.slab_key(orphan_txg, 0); + let orphan = h.data.get_blob(&orphan_key, None).await.unwrap().body; + drop(store); + + let store = h.open().await.unwrap(); + let objid = store.reserve_objid().await.unwrap(); + let root = store + .commit(BTreeMap::from([(objid, file(objid, 3))])) + .await + .unwrap(); + + assert!( + root.txg > orphan_txg, + "commit reused transaction group {} (orphan at {orphan_txg})", + root.txg + ); + // And the orphan is untouched, not overwritten. + assert_eq!( + h.data.get_blob(&orphan_key, None).await.unwrap().body, + orphan + ); + } + + /// A transaction dropped without committing consumed a number that no + /// root records. The next commit has to claim before it seals, or a + /// crash after its slabs would leave that number two past the tip — + /// exactly where the next session would start. + #[tokio::test] + async fn an_abandoned_transaction_forces_a_claim() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let before = store.root().await.unwrap(); + + let txn = store.begin().await.unwrap(); + let abandoned = txn.txg(); + drop(txn); + + let objid = store.reserve_objid().await.unwrap(); + let root = store + .commit(BTreeMap::from([(objid, file(objid, 1))])) + .await + .unwrap(); + + // One claim root in between, naming a number past the abandoned one. + assert_eq!(root.seq, before.seq + 2); + let claim = store.roots.load_snapshot(before.seq + 1).await.unwrap(); + assert!(claim.txg > abandoned); + assert_eq!(root.txg, claim.txg + 1); + } + + /// A commit that dies short of its root has already sealed blocks under + /// its transaction group. The retry must not only skip that number but + /// put it on record first, so a crash before the retry lands does not + /// hand it to the next session. + #[tokio::test] + async fn a_transaction_group_is_burned_even_when_the_commit_fails() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let before = store.root().await.unwrap(); + + h.data.fail_next_put(FsError::IoTimeout); + let objid = store.reserve_objid().await.unwrap(); + assert!(matches!( + store + .commit(BTreeMap::from([(objid, file(objid, 1))])) + .await, + Err(FsError::IoTimeout) + )); + let failed_txg = store.state.lock().await.next_txg - 1; + assert!(store.poison().await.is_none(), "a timeout is retryable"); + + let root = store + .commit(BTreeMap::from([(objid, file(objid, 1))])) + .await + .unwrap(); + assert!(root.txg > failed_txg, "reused {}", root.txg); + + // The retry re-claimed first: the chain now holds a root past the + // burned number, so no later session can start on it. + let claim = store.roots.load_snapshot(root.seq - 1).await.unwrap(); + assert_eq!(claim.seq, before.seq + 1); + assert!(claim.txg > failed_txg); + assert_eq!(root.txg, claim.txg + 1); + } + + /// The number a session seals under comes from the locked root chain, + /// never from what happens to be in the data bucket. A host that deletes + /// an orphan gains nothing. + #[tokio::test] + async fn a_new_session_starts_past_what_the_last_one_could_have_sealed() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let tip = store.root().await.unwrap().txg; + + // Claim, then die before writing anything: the data bucket has no + // trace of the number this session was about to use. + let txn = store.begin().await.unwrap(); + let burned = txn.txg(); + drop(txn); + drop(store); + assert!(h + .data + .keys() + .iter() + .all(|k| !k.contains(&format!("{burned:016x}")))); + + let store = h.open().await.unwrap(); + let objid = store.reserve_objid().await.unwrap(); + let root = store + .commit(BTreeMap::from([(objid, file(objid, 1))])) + .await + .unwrap(); + assert!( + root.txg > burned, + "{} <= {burned} (tip was {tip})", + root.txg + ); + } + + /// Two mounts at one tip would allocate the same slab names and the same + /// nonces. The claim decides between them before either has ciphertext. + #[tokio::test] + async fn a_mount_that_loses_the_claim_never_seals_a_block() { + let h = Harness::new(); + let a = h.open().await.unwrap(); + let b = h.open().await.unwrap(); + + let objid = a.reserve_objid().await.unwrap(); + a.commit(BTreeMap::from([(objid, file(objid, 1))])) + .await + .unwrap(); + + let slabs_before = h.data.object_count(); + let objid = b.reserve_objid().await.unwrap(); + assert!(matches!( + b.commit(BTreeMap::from([(objid, file(objid, 2))])).await, + Err(FsError::Conflict) + )); + assert!(matches!(b.poison().await, Some(FsError::Conflict))); + assert_eq!( + h.data.object_count(), + slabs_before, + "the loser wrote a slab" + ); + + // And the winner's data is intact for the next session. + let c = h.open().await.unwrap(); + let root = c.root().await.unwrap(); + assert_eq!(root.seq, a.root().await.unwrap().seq); + } + + /// The same race with a delete marker laid over the winner's claim. The + /// marker lets the loser's conditional create through, because + /// `If-None-Match: *` sees only the current version — and if that were + /// the end of it, both mounts would seal under one transaction group: the + /// same nonces under the same key, into the same slab names. + #[tokio::test] + async fn a_delete_marker_does_not_hand_the_claim_to_a_second_mount() { + let h = Harness::new(); + drop(h.open().await.unwrap()); + let a = h.open().await.unwrap(); + let b = h.open().await.unwrap(); + + let _sealing = a.begin().await.unwrap(); + h.roots.delete_blob(&h.config.root_key(1)).await.unwrap(); + + assert!(matches!(b.begin().await, Err(FsError::Conflict))); + assert!(matches!(b.poison().await, Some(FsError::Conflict))); + } + + // ---- poisoning --------------------------------------------------------- + + /// Losing the race for a sequence number is fatal. Retrying would reuse + /// the transaction group and repeat every nonce in it. + #[tokio::test] + async fn losing_the_sequence_race_poisons_the_mount() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let genesis = store.root().await.unwrap(); + + let rs = RootStore::new(h.roots.clone(), h.keys.clone(), h.config.clone()); + let stolen = RootRecord::seal( + &h.keys, + 1, + Some(&genesis), + 999, + 0, + genesis.meta_dnode.clone(), + 2, + ) + .unwrap(); + rs.publish(&stolen).await.unwrap(); + + let objid = store.reserve_objid().await.unwrap(); + assert!(matches!( + store + .commit(BTreeMap::from([(objid, file(objid, 1))])) + .await, + Err(FsError::Conflict) + )); + + assert!(matches!(store.poison().await, Some(FsError::Conflict))); + // Everything afterwards fails, including reads: we no longer know what + // is committed. + assert!(store.root().await.is_err()); + assert!(store.objset().await.is_err()); + assert!(store.commit(BTreeMap::new()).await.is_err()); + } + + #[tokio::test] + async fn a_healthy_mount_is_not_poisoned() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let objid = store.reserve_objid().await.unwrap(); + store + .commit(BTreeMap::from([(objid, file(objid, 1))])) + .await + .unwrap(); + assert!(store.poison().await.is_none()); + } + + // ---- genesis is not a fallback ---------------------------------------- + + /// The whole point of the split. An empty store used to mean "make me a + /// filesystem"; a store that answers "nothing" is now an error, because + /// only the caller can tell a genuinely new filesystem from one that has + /// been hidden. + #[tokio::test] + async fn mounting_an_empty_store_refuses_rather_than_formatting() { + let h = Harness::new(); + assert!(matches!(h.mount(None).await, Err(FsError::NoFilesystem))); + + // And it really was left empty — the failed mount must not have + // created anything on its way out. + assert!(matches!(h.mount(None).await, Err(FsError::NoFilesystem))); + } + + /// The other direction: a caller that has misjudged which mode it is in + /// must not be able to overwrite a filesystem it should have mounted. + #[tokio::test] + async fn creating_over_an_existing_filesystem_is_refused() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let objid = store.reserve_objid().await.unwrap(); + store + .commit(BTreeMap::from([(objid, file(objid, 7))])) + .await + .unwrap(); + let before = store.root().await.unwrap(); + drop(store); + + assert!(matches!( + Store::create( + h.data.clone(), + h.roots.clone(), + h.keys.clone(), + h.config.clone() + ) + .await, + Err(FsError::AlreadyExists) + )); + + // The refusal left the existing filesystem exactly as it was. + let after = h.mount(None).await.unwrap().root().await.unwrap(); + assert_eq!(after, before); + } + + #[tokio::test] + async fn a_created_filesystem_mounts_back() { + let h = Harness::new(); + let genesis = Store::create( + h.data.clone(), + h.roots.clone(), + h.keys.clone(), + h.config.clone(), + ) + .await + .unwrap() + .root() + .await + .unwrap(); + + assert_eq!(genesis.seq, 0); + assert!( + genesis.prev_root_hash.is_zero(), + "genesis has no predecessor" + ); + assert_eq!(h.mount(None).await.unwrap().root().await.unwrap(), genesis); + } + + // ---- rollback ---------------------------------------------------------- + + #[tokio::test] + async fn a_mount_floor_rejects_a_rewound_store() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + for i in 0..3u64 { + let objid = store.reserve_objid().await.unwrap(); + store + .commit(BTreeMap::from([(objid, file(objid, i))])) + .await + .unwrap(); + } + drop(store); + + // Mounting with a floor at or below the real tip is fine. + assert!(h.mount(Some(3)).await.is_ok()); + + // A floor above it means the store is behind where we know it to be. + assert!(matches!( + h.mount(Some(4)).await, + Err(FsError::Rollback { + expected: 4, + found: 3 + }) + )); + } + + /// The end-to-end rollback story: an adversary who can write to the bucket + /// cannot make a mount accept an older state. + /// + /// Not because the newer root cannot be deleted — it can be *hidden*, and + /// this test does exactly that — but because a hidden version is still a + /// version, and tip discovery reads the retained one. The sequence is + /// checked against the key on top of that. + #[tokio::test] + async fn an_adversary_cannot_rewind_the_filesystem() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let objid = store.reserve_objid().await.unwrap(); + store + .commit(BTreeMap::from([(objid, file(objid, 111))])) + .await + .unwrap(); + let old_root = store.root().await.unwrap(); + store + .commit(BTreeMap::from([(objid, file(objid, 222))])) + .await + .unwrap(); + drop(store); + + // Hide the newest root. S3 allows this: the delete marker goes on top + // and the retained version stays underneath, undeletable. + h.roots + .delete_blob(&h.config.root_key(2)) + .await + .expect("a delete marker is a legal write"); + assert!( + matches!( + h.roots.get_blob(&h.config.root_key(2), None).await, + Err(FsError::NotFound) + ), + "an ordinary read now reports the tip missing — which is the whole \ + attack, and why tip discovery must not use one" + ); + + // Try to replay the older root over the newer key: also refused, and + // even if it landed, its `seq` would not match the key. + assert!(matches!( + h.roots + .put_blob(PutBlobInput::new( + h.config.root_key(2), + Bytes::from(old_root.encode().unwrap()) + )) + .await, + Err(FsError::AccessDenied) + )); + + // The filesystem still mounts at the newest state. + let store = h.open().await.unwrap(); + assert_eq!(store.root().await.unwrap().seq, 2); + let objset = store.objset().await.unwrap(); + assert_eq!( + objset + .get_allocated(store.blocks(), objid) + .await + .unwrap() + .size, + 222 + ); + } + + #[tokio::test] + async fn a_replayed_root_at_the_wrong_key_is_refused() { + let h = Harness::new(); + let store = h.open().await.unwrap(); + let objid = store.reserve_objid().await.unwrap(); + store + .commit(BTreeMap::from([(objid, file(objid, 1))])) + .await + .unwrap(); + let old = store.root().await.unwrap(); + drop(store); + + // Plant root 1's bytes at key 2, which no one has claimed yet. + h.roots + .put_blob_if_not_exists(PutBlobInput::new( + h.config.root_key(2), + Bytes::from(old.encode().unwrap()), + )) + .await + .unwrap(); + + // Tip discovery finds key 2; verification rejects it, and the mount + // fails closed rather than silently falling back to key 1. + assert!(matches!( + h.open().await, + Err(FsError::Integrity("root: sequence does not match its key")) + )); + } +} diff --git a/crates/s3fs-core/tests/minio_integration.rs b/crates/s3fs-core/tests/minio_integration.rs index a3b636b..35d6c61 100644 --- a/crates/s3fs-core/tests/minio_integration.rs +++ b/crates/s3fs-core/tests/minio_integration.rs @@ -1,9 +1,10 @@ //! Integration tests for `AwsS3Backend` against a MinIO container. //! //! Every test is `#[ignore]`d so default `cargo test` skips them. Run with -//! Docker present: +//! Docker present, after building the image they run on: //! //! ```bash +//! scripts/minio-image.sh //! cargo test -p s3fs-core --features aws --test minio_integration -- --ignored //! ``` //! @@ -19,25 +20,29 @@ use std::time::Duration; use bytes::Bytes; use testcontainers::runners::AsyncRunner; -use testcontainers::ContainerAsync; +use testcontainers::{ContainerAsync, ImageExt}; use testcontainers_modules::minio::MinIO; use s3fs_core::backend::{ AwsS3Backend, AwsS3BackendConfig, Backend, CompletedPart, ListBlobsInput, PutBlobInput, }; -use s3fs_core::config::PartSchedule; +use s3fs_core::backend::{ObjectLock, ObjectLockMode}; use s3fs_core::errors::FsError; use s3fs_core::fs::OpenFlags; -use s3fs_core::{Config, Fs}; +use s3fs_core::{Config, Fs, MasterSecret}; /// Start a MinIO container, build an `AwsS3Backend` pointed at it, and create /// the bucket. The returned `ContainerAsync` MUST stay in scope for the /// lifetime of the test — when it drops, the container is killed. +/// +/// The module's own release, under the name `scripts/minio-image.sh` builds it +/// as: MinIO's published images can no longer be pulled. async fn fresh_minio_with_bucket(bucket: &str) -> (ContainerAsync, AwsS3Backend) { let container = MinIO::default() + .with_name("enclave-runtime/minio") .start() .await - .expect("MinIO container start"); + .expect("MinIO container start; is the image built? scripts/minio-image.sh"); let port = container .get_host_port_ipv4(9000) .await @@ -69,6 +74,21 @@ async fn fresh_minio_with_bucket(bucket: &str) -> (ContainerAsync, AwsS3B (container, backend) } +/// A second backend against the same container, for the roots bucket. +async fn extra_bucket(backend: &AwsS3Backend, bucket: &str, object_lock: bool) -> AwsS3Backend { + let mut req = backend.client().create_bucket().bucket(bucket); + if object_lock { + req = req.object_lock_enabled_for_bucket(true); + } + req.send().await.expect("create bucket"); + + let mut cfg = backend.config().clone(); + cfg.bucket = bucket.to_string(); + AwsS3Backend::connect_unchecked(cfg) + .await + .expect("connect to second bucket") +} + // ---------- Backend-level tests ---------- #[tokio::test(flavor = "multi_thread")] @@ -84,6 +104,7 @@ async fn put_head_get_delete_round_trip() { body: Bytes::from_static(b"hello"), metadata: metadata.clone(), content_type: Some("text/plain".into()), + object_lock: None, }) .await .unwrap(); @@ -113,6 +134,7 @@ async fn get_with_byte_range() { body: Bytes::from_static(b"0123456789"), metadata: HashMap::new(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -129,6 +151,7 @@ async fn put_blob_if_not_exists_returns_already_exists_on_collision() { body: Bytes::from_static(b"v"), metadata: HashMap::new(), content_type: None, + object_lock: None, }; backend.put_blob_if_not_exists(mk()).await.unwrap(); assert!(matches!( @@ -148,6 +171,7 @@ async fn list_with_delimiter_separates_items_and_prefixes() { body: Bytes::from_static(b""), metadata: HashMap::new(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -184,6 +208,7 @@ async fn multipart_lifecycle_with_upload_part_copy() { body: Bytes::from(src_body.clone()), metadata: HashMap::new(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -195,6 +220,7 @@ async fn multipart_lifecycle_with_upload_part_copy() { body: Bytes::new(), metadata: HashMap::new(), content_type: None, + object_lock: None, }) .await .unwrap(); @@ -238,158 +264,308 @@ async fn multipart_lifecycle_with_upload_part_copy() { // ---------- Fs end-to-end tests ---------- +// ---------- Object Lock ---------- + +/// What Object Lock COMPLIANCE actually buys, against a real S3 implementation. +/// +/// It buys indestructibility, not invisibility. `DeleteObject` without a +/// version id **succeeds** and writes a delete marker; an ordinary `GetObject` +/// then answers `NoSuchKey`. The version underneath cannot be removed — that +/// fails with *"Object is WORM protected"* — and +/// [`Backend::get_retained_blob`] is how the engine reaches it. +/// +/// This test used to assert the delete was refused. It was wrong, and it is +/// where the whole delete-marker problem was found. #[tokio::test(flavor = "multi_thread")] -#[ignore = "requires Docker"] -async fn fs_small_file_round_trip() { - let (_c, backend) = fresh_minio_with_bucket("fs-small").await; - let cfg = Config::builder() - .single_part_threshold(5 * 1024 * 1024) - .build(); - let fs = Fs::new(Arc::new(backend) as Arc, Arc::new(cfg)); - - let h = fs - .open( - "hello.txt", - OpenFlags { - read: true, - write: true, - create: true, - ..Default::default() - }, - ) +#[ignore = "requires Docker; run with --features aws -- --ignored"] +async fn a_retained_root_can_be_hidden_but_not_destroyed() { + let (_c, data) = fresh_minio_with_bucket("data-lock").await; + let roots = extra_bucket(&data, "roots-lock", true).await; + + // A minute, not a decade: the container is thrown away, but a real + // COMPLIANCE retention would make the bucket itself undeletable for the + // full term, which is not something a test should create. + let lock = ObjectLock { + mode: ObjectLockMode::Compliance, + retain_until: std::time::SystemTime::now() + Duration::from_secs(60), + }; + let input = PutBlobInput::new("roots/0000000000000000", Bytes::from_static(b"anchor")) + .with_object_lock(lock); + roots.put_blob_if_not_exists(input).await.unwrap(); + + // S3 permits this. Asserting otherwise is how the fake came to be stronger + // than the thing it stands for. + roots + .delete_blob("roots/0000000000000000") .await - .unwrap(); - fs.pwrite(&h, 0, b"hello, minio").await.unwrap(); - fs.sync(&h).await.unwrap(); - fs.close(&h).await.unwrap(); + .expect("a delete marker is a legal write"); + + // Hidden from the ordinary read the engine used to use... + assert!( + matches!( + roots.get_blob("roots/0000000000000000", None).await, + Err(FsError::NotFound) + ), + "a delete marker must hide the current version" + ); + + // ...and still there, which is the guarantee that survives. + assert_eq!( + roots + .get_retained_blob("roots/0000000000000000") + .await + .expect("the retained version is reachable past the marker") + .body, + Bytes::from_static(b"anchor") + ); + + // Why hiding is dangerous where absence carries meaning: the conditional + // PUT that serves as "am I the first here?" is satisfied again. + roots + .put_blob_if_not_exists(PutBlobInput::new( + "roots/0000000000000000", + Bytes::from_static(b"forged"), + )) + .await + .expect("If-None-Match is satisfied by a delete marker"); - // Re-open and read back. - let h2 = fs.open("hello.txt", OpenFlags::read_only()).await.unwrap(); - let body = fs.pread(&h2, 0, 100).await.unwrap(); - assert_eq!(&body[..], b"hello, minio"); - fs.close(&h2).await.unwrap(); + // But the pinned version is untouched underneath, so reading the retained + // version still gets the real one. + assert_eq!( + roots + .get_retained_blob("roots/0000000000000000") + .await + .unwrap() + .body, + Bytes::from_static(b"anchor"), + "the retained version must outrank anything written over the marker" + ); } -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires Docker"] -async fn fs_in_place_update_uses_copy_for_unchanged_parts() { - // The load-bearing claim end-to-end: a 15 MiB file (3 parts of 5 MiB), - // modify only part 1 (bytes 5MiB..10MiB), sync, verify content. The Fs - // commit path should issue UploadPartCopy for parts 0 and 2 — no full - // re-upload of unchanged data. - let (_c, backend) = fresh_minio_with_bucket("fs-rmw").await; - let cfg = Config::builder() - .part_schedule(PartSchedule { - tiers: vec![(5 * 1024 * 1024, 1000)], - }) - .single_part_threshold(5 * 1024 * 1024) - .max_parallel_parts(4) - .max_parallel_copy(4) - .max_merge_copy_bytes(128 * 1024 * 1024) - .memory_limit_bytes(64 * 1024 * 1024) - .build(); - let backend_arc = Arc::new(backend); - let fs = Fs::new(backend_arc.clone() as Arc, Arc::new(cfg)); - - // Pre-populate "big" with 15 MiB of recognisable bytes. - let total: usize = 15 * 1024 * 1024; - let mut src = vec![0u8; total]; - for (i, b) in src.iter_mut().enumerate() { - *b = (i % 251) as u8; - } - backend_arc - .put_blob(PutBlobInput { - key: "big".into(), - body: Bytes::from(src.clone()), - metadata: HashMap::new(), - content_type: None, - }) - .await - .unwrap(); +// ---------- Engine-level tests ---------- - // Open and overwrite part 1 (bytes 5MiB..10MiB). - let h = fs - .open( - "big", - OpenFlags { - read: true, - write: true, - ..Default::default() - }, +const TEST_MASTER: [u8; 32] = [0x42; 32]; +const TEST_FS_ID: [u8; 16] = [0x11; 16]; + +fn engine_config() -> Config { + Config::builder() + .record_size(64 * 1024) + // Retention is exercised separately; leaving it off keeps these tests + // able to tear their buckets down. + .root_retention(None) + .build() +} + +/// Create on first use, mount thereafter — several tests here remount a bucket +/// to prove the state survived, and `Fs::mount` no longer formats an empty +/// store, so wanting a filesystem has to be said out loud. +async fn mount(data: &AwsS3Backend, roots: &AwsS3Backend) -> Arc { + let data = Arc::new(data.clone()) as Arc; + let roots = Arc::new(roots.clone()) as Arc; + match Fs::mount( + data.clone(), + roots.clone(), + &MasterSecret::from_bytes(TEST_MASTER), + TEST_FS_ID, + Arc::new(engine_config()), + None, + ) + .await + { + Err(s3fs_core::FsError::NoFilesystem) => Fs::create( + data, + roots, + &MasterSecret::from_bytes(TEST_MASTER), + TEST_FS_ID, + Arc::new(engine_config()), ) .await - .unwrap(); - let new_chunk = vec![0xAB; 5 * 1024 * 1024]; - fs.pwrite(&h, 5 * 1024 * 1024, &new_chunk).await.unwrap(); - fs.sync(&h).await.unwrap(); - fs.close(&h).await.unwrap(); - - // Verify the resulting object byte-for-byte. - let g = backend_arc.get_blob("big", None).await.unwrap(); - assert_eq!(g.body.len(), total); - for i in 0..total { - let expected = if (5 * 1024 * 1024..10 * 1024 * 1024).contains(&i) { - 0xAB - } else { - (i % 251) as u8 - }; - assert_eq!(g.body[i], expected, "byte {i}"); + .expect("create"), + other => other.expect("mount"), } } #[tokio::test(flavor = "multi_thread")] -#[ignore = "requires Docker"] -async fn fs_mkdir_and_listing() { - let (_c, backend) = fresh_minio_with_bucket("fs-mkdir").await; - let cfg = Config::default(); - let fs = Fs::new(Arc::new(backend) as Arc, Arc::new(cfg)); - - let root = fs.root(); - fs.mkdir(&root, "subdir").await.unwrap(); - let h = fs - .open( - "f.txt", - OpenFlags { - write: true, - create: true, - ..Default::default() - }, - ) +#[ignore = "requires Docker; run with --features aws -- --ignored"] +async fn end_to_end_write_read_and_remount() { + let (_c, data) = fresh_minio_with_bucket("data-e2e").await; + let roots = extra_bucket(&data, "roots-e2e", false).await; + + let payload: Vec = (0..300_000u32).map(|i| (i % 251) as u8).collect(); + { + let fs = mount(&data, &roots).await; + let root = fs.root(); + fs.mkdir(&root, "dir").await.unwrap(); + let h = fs.open("/dir/file", OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&h, 0, &payload).await.unwrap(); + fs.close(&h).await.unwrap(); + } + + // A separate mount, reading only what the anchored root records commit. + let fs = mount(&data, &roots).await; + let h = fs.open("/dir/file", OpenFlags::read_only()).await.unwrap(); + let got = fs.pread(&h, 0, payload.len()).await.unwrap(); + assert_eq!(got.len(), payload.len()); + assert_eq!(got.as_ref(), payload.as_slice()); + + let names: Vec<_> = fs + .read_dir(&fs.root()) .await - .unwrap(); - fs.pwrite(&h, 0, b"x").await.unwrap(); - fs.sync(&h).await.unwrap(); + .unwrap() + .into_iter() + .map(|e| e.name) + .collect(); + assert_eq!(names, vec!["dir"]); +} + +#[tokio::test(flavor = "multi_thread")] +#[ignore = "requires Docker; run with --features aws -- --ignored"] +async fn a_commit_costs_a_handful_of_objects() { + let (_c, data) = fresh_minio_with_bucket("data-cost").await; + let roots = extra_bucket(&data, "roots-cost", false).await; + let fs = mount(&data, &roots).await; + + let before = count_keys(&data, "slabs/").await; + let h = fs.open("/big", OpenFlags::create_new()).await.unwrap(); + // 64 records at the configured record size, committed as one group. + fs.pwrite(&h, 0, &vec![7u8; 64 * 64 * 1024]).await.unwrap(); fs.close(&h).await.unwrap(); + let after = count_keys(&data, "slabs/").await; + + // The point of packing blocks into slabs: many dirty blocks, few PUTs. + // Addressing blocks by content hash would have cost 64 objects here. + assert!( + after - before <= 4, + "expected a handful of slab objects, got {}", + after - before + ); +} - let snap = fs.read_dir(&root).await.unwrap(); - let names: Vec<_> = snap.iter().map(|e| e.name.as_str()).collect(); - assert!(names.contains(&"subdir")); - assert!(names.contains(&"f.txt")); +#[tokio::test(flavor = "multi_thread")] +#[ignore = "requires Docker; run with --features aws -- --ignored"] +async fn tampering_with_a_slab_is_detected() { + let (_c, data) = fresh_minio_with_bucket("data-tamper").await; + let roots = extra_bucket(&data, "roots-tamper", false).await; + + { + let fs = mount(&data, &roots).await; + let h = fs.open("/f", OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&h, 0, &vec![0x5au8; 100_000]).await.unwrap(); + fs.close(&h).await.unwrap(); + } + + // Rewrite every slab with a flipped byte, as a bucket operator could. + for key in list_keys(&data, "slabs/").await { + let body = data.get_blob(&key, None).await.unwrap().body; + let mut body = body.to_vec(); + body[0] ^= 0xff; + data.put_blob(PutBlobInput::new(key, Bytes::from(body))) + .await + .unwrap(); + } + + let fs = mount(&data, &roots).await; + let result = async { + let h = fs.open("/f", OpenFlags::read_only()).await?; + fs.pread(&h, 0, 100_000).await + } + .await; + assert!( + matches!(result, Err(FsError::Integrity(_))), + "expected an integrity failure, got {result:?}" + ); } #[tokio::test(flavor = "multi_thread")] -#[ignore = "requires Docker"] -async fn fs_o_excl_collision_returns_already_exists() { - let (_c, backend) = fresh_minio_with_bucket("fs-excl").await; - let cfg = Config::default(); - let fs = Fs::new(Arc::new(backend) as Arc, Arc::new(cfg)); - - let _h1 = fs.open("k", OpenFlags::create_new()).await.unwrap(); - let r = fs.open("k", OpenFlags::create_new()).await; - assert!(matches!(r, Err(FsError::AlreadyExists))); +#[ignore = "requires Docker; run with --features aws -- --ignored"] +async fn a_mount_floor_above_the_tip_is_refused() { + let (_c, data) = fresh_minio_with_bucket("data-floor").await; + let roots = extra_bucket(&data, "roots-floor", false).await; + { + let fs = mount(&data, &roots).await; + fs.mkdir(&fs.root(), "a").await.unwrap(); + } + + let result = Fs::mount( + Arc::new(data.clone()) as Arc, + Arc::new(roots.clone()) as Arc, + &MasterSecret::from_bytes(TEST_MASTER), + TEST_FS_ID, + Arc::new(engine_config()), + Some(9999), + ) + .await; + assert!(matches!(result, Err(FsError::Rollback { .. }))); } +/// Two mounts at one tip, and a delete marker laid over the first one's claim +/// before the second makes its own. Real S3 accepts the second conditional PUT +/// — the marker is the current version — and without the read-back of the +/// retained version the second mount seals under the first one's transaction +/// group: the same nonces, and the same slab names, so its upload overwrites +/// what the first had committed. It still loses the *next* sequence, so its +/// refusal alone proves little. The first mount's data surviving is the proof. +/// +/// Versioning is what keeps that claim beneath the marker; retention, tested +/// above, is what stops anyone deleting the version itself. #[tokio::test(flavor = "multi_thread")] -#[ignore = "requires Docker"] -async fn fs_symlink_round_trip() { - let (_c, backend) = fresh_minio_with_bucket("fs-sym").await; - let cfg = Config::default(); - let fs = Fs::new(Arc::new(backend) as Arc, Arc::new(cfg)); - - let root = fs.root(); - fs.symlink_at(&root, "link", "target.txt").await.unwrap(); - assert_eq!(fs.readlink_at(&root, "link").await.unwrap(), "target.txt"); - // Re-creating the same symlink should fail with AlreadyExists. - let r = fs.symlink_at(&root, "link", "other.txt").await; - assert!(matches!(r, Err(FsError::AlreadyExists))); +#[ignore = "requires Docker; run with --features aws -- --ignored"] +async fn a_delete_marker_does_not_let_a_second_mount_claim_a_sequence() { + let (_c, data) = fresh_minio_with_bucket("data-claim").await; + let roots = extra_bucket(&data, "roots-claim", true).await; + drop(mount(&data, &roots).await); + let a = mount(&data, &roots).await; + let b = mount(&data, &roots).await; + + a.mkdir(&a.root(), "from-a").await.unwrap(); + roots + .delete_blob(&engine_config().store.root_key(1)) + .await + .expect("a delete marker is a legal write"); + + assert!(matches!( + b.mkdir(&b.root(), "from-b").await, + Err(FsError::Conflict) + )); + // Reads back the blocks it committed, which the loser would have + // overwritten. + a.mkdir(&a.root(), "still-a").await.unwrap(); + + // The loser's claim sits above the marker, but it is not the one a mount + // reads: the winner's history is the filesystem. + let c = mount(&data, &roots).await; + let mut names: Vec<_> = c + .read_dir(&c.root()) + .await + .unwrap() + .into_iter() + .map(|e| e.name) + .collect(); + names.sort(); + assert_eq!(names, vec!["from-a", "still-a"]); +} + +async fn list_keys(backend: &AwsS3Backend, prefix: &str) -> Vec { + let mut out = Vec::new(); + let mut token: Option = None; + loop { + let page = backend + .list_blobs(ListBlobsInput { + prefix, + continuation_token: token.as_deref(), + ..Default::default() + }) + .await + .unwrap(); + out.extend(page.items.into_iter().map(|i| i.key)); + match page.next_continuation_token { + Some(t) if page.is_truncated => token = Some(t), + _ => break, + } + } + out +} + +async fn count_keys(backend: &AwsS3Backend, prefix: &str) -> usize { + list_keys(backend, prefix).await.len() } diff --git a/crates/s3fs-runner/Cargo.toml b/crates/s3fs-runner/Cargo.toml deleted file mode 100644 index e1e9a00..0000000 --- a/crates/s3fs-runner/Cargo.toml +++ /dev/null @@ -1,28 +0,0 @@ -[package] -name = "s3fs-runner" -version.workspace = true -edition.workspace = true -rust-version.workspace = true -license.workspace = true -repository.workspace = true -description = "Wasmtime runner that loads a Wasm component and exposes wasi:filesystem backed by S3" - -[[bin]] -name = "s3fs-runner" -path = "src/main.rs" - -[dependencies] -s3fs-core = { path = "../s3fs-core", features = ["aws"] } -s3fs-wasmtime = { path = "../s3fs-wasmtime" } - -# Runner needs the compiler (cranelift) plus runtime; the workspace base dep -# is runtime-only so the library crate stays light. -wasmtime = { version = "=44.0.0", features = ["component-model", "async", "runtime", "cranelift", "std"] } -wasmtime-wasi = { workspace = true } -wasmtime-wasi-io = "=44.0.0" - -clap = { version = "4", features = ["derive", "env"] } -tokio = { workspace = true, features = ["rt-multi-thread", "macros", "signal"] } -anyhow = { workspace = true } -tracing = { workspace = true } -tracing-subscriber = { version = "0.3", features = ["env-filter", "fmt"] } diff --git a/crates/s3fs-runner/src/main.rs b/crates/s3fs-runner/src/main.rs deleted file mode 100644 index c076c33..0000000 --- a/crates/s3fs-runner/src/main.rs +++ /dev/null @@ -1,256 +0,0 @@ -//! `s3fs-runner` — load a `wasi:cli/command` component and run it with -//! `wasi:filesystem` backed by S3 (or S3-compatible storage like MinIO). -//! -//! The full WASI surface is provided: `wasi:io`, `wasi:cli`, `wasi:clocks`, -//! `wasi:random`, and `wasi:sockets` come from `wasmtime-wasi`; only -//! `wasi:filesystem` is taken over by `s3fs-wasmtime` and routed to an -//! `s3fs_core::Fs` over an `AwsS3Backend`. - -use std::path::PathBuf; -use std::sync::Arc; -use std::time::Duration; - -use anyhow::{Context, Result as AnyResult}; -use clap::Parser; -use s3fs_core::backend::{AwsS3Backend, AwsS3BackendConfig, Backend}; -use s3fs_core::Fs; -use s3fs_wasmtime::{S3FsCtxView, S3WasiView}; -use wasmtime::component::{Component, HasData, Linker, ResourceTable}; -use wasmtime::{Config, Engine, Store}; -use wasmtime_wasi::cli::{WasiCli, WasiCliView}; -use wasmtime_wasi::clocks::{WasiClocks, WasiClocksView}; -use wasmtime_wasi::random::{WasiRandom, WasiRandomView}; -use wasmtime_wasi::sockets::{WasiSockets, WasiSocketsView}; -use wasmtime_wasi::{WasiCtx, WasiCtxBuilder, WasiCtxView, WasiView}; - -#[derive(Parser, Debug)] -#[command( - version, - about = "Run a Wasm component with wasi:filesystem backed by S3" -)] -struct Cli { - /// S3 bucket name. - #[arg(long, env = "S3FS_BUCKET")] - bucket: String, - - /// AWS region. - #[arg(long, env = "AWS_REGION", default_value = "us-east-1")] - region: String, - - /// Endpoint override (e.g. http://127.0.0.1:9000 for MinIO). - #[arg(long, env = "S3FS_ENDPOINT")] - endpoint: Option, - - #[arg(long, env = "AWS_ACCESS_KEY_ID")] - access_key_id: Option, - - #[arg(long, env = "AWS_SECRET_ACCESS_KEY")] - secret_access_key: Option, - - #[arg(long, env = "AWS_SESSION_TOKEN")] - session_token: Option, - - /// Use path-style addressing (required for MinIO and many S3-compatibles). - #[arg(long, env = "S3FS_FORCE_PATH_STYLE")] - force_path_style: bool, - - /// Optional key prefix inside the bucket. All paths are scoped under this. - #[arg(long, env = "S3FS_BUCKET_PREFIX", default_value = "")] - bucket_prefix: String, - - /// Path the guest sees as its preopen root (default `/`). - #[arg(long, env = "S3FS_MOUNT_PATH", default_value = "/")] - mount_path: String, - - /// Skip the `HeadBucket` startup probe. - #[arg(long)] - skip_bucket_probe: bool, - - /// Path to the `.wasm` component file. - #[arg(long, short = 'c')] - component: PathBuf, - - /// Arguments to pass to the guest as `wasi:cli/environment.get-arguments()`. - #[arg(last = true)] - guest_args: Vec, -} - -/// The store-data type. Implements both `WasiView` (so `wasmtime-wasi` can -/// serve `wasi:io`/`wasi:cli`/etc.) and `S3WasiView` (so `s3fs-wasmtime` can -/// serve `wasi:filesystem`). -struct State { - wasi: WasiCtx, - table: ResourceTable, - fs: Arc, -} - -impl WasiView for State { - fn ctx(&mut self) -> WasiCtxView<'_> { - WasiCtxView { - ctx: &mut self.wasi, - table: &mut self.table, - } - } -} - -impl S3WasiView for State { - fn s3fs_view(&mut self) -> S3FsCtxView<'_> { - // Share the SAME ResourceTable instance with wasmtime-wasi so stream - // resources we push are visible to wasmtime-wasi's stream methods. - S3FsCtxView { - fs: &self.fs, - table: &mut self.table, - } - } -} - -/// Marker for `wasi:io` interfaces — they need `&mut ResourceTable`. -struct HasIo; -impl HasData for HasIo { - type Data<'a> = &'a mut ResourceTable; -} - -/// Add every `wasmtime-wasi` interface to the linker EXCEPT `wasi:filesystem`, -/// which `s3fs-wasmtime` will provide instead. -/// -/// Body cloned from `wasmtime_wasi::p2::add_to_linker_with_options_async`, -/// minus the two `filesystem::*` lines. -fn add_wasi_minus_filesystem(linker: &mut Linker) -> AnyResult<()> { - use wasmtime_wasi::p2::bindings::{cli, clocks, random, sockets}; - use wasmtime_wasi_io::bindings::wasi::io; - let l = linker; - let options = wasmtime_wasi::p2::bindings::LinkOptions::default(); - - // wasi:io (async) - io::error::add_to_linker::(l, |t| t.ctx().table)?; - io::poll::add_to_linker::(l, |t| t.ctx().table)?; - io::streams::add_to_linker::(l, |t| t.ctx().table)?; - - // sockets (async) - sockets::tcp::add_to_linker::(l, State::sockets)?; - sockets::udp::add_to_linker::(l, State::sockets)?; - - // clocks - clocks::wall_clock::add_to_linker::(l, State::clocks)?; - clocks::monotonic_clock::add_to_linker::(l, State::clocks)?; - - // random - random::random::add_to_linker::(l, State::random)?; - random::insecure::add_to_linker::(l, State::random)?; - random::insecure_seed::add_to_linker::(l, State::random)?; - - // cli - cli::exit::add_to_linker::(l, &(&options).into(), State::cli)?; - cli::environment::add_to_linker::(l, State::cli)?; - cli::stdin::add_to_linker::(l, State::cli)?; - cli::stdout::add_to_linker::(l, State::cli)?; - cli::stderr::add_to_linker::(l, State::cli)?; - cli::terminal_input::add_to_linker::(l, State::cli)?; - cli::terminal_output::add_to_linker::(l, State::cli)?; - cli::terminal_stdin::add_to_linker::(l, State::cli)?; - cli::terminal_stdout::add_to_linker::(l, State::cli)?; - cli::terminal_stderr::add_to_linker::(l, State::cli)?; - - // sockets (non-async) - sockets::tcp_create_socket::add_to_linker::(l, State::sockets)?; - sockets::udp_create_socket::add_to_linker::(l, State::sockets)?; - sockets::instance_network::add_to_linker::(l, State::sockets)?; - sockets::network::add_to_linker::(l, &(&options).into(), State::sockets)?; - sockets::ip_name_lookup::add_to_linker::(l, State::sockets)?; - - Ok(()) -} - -#[tokio::main] -async fn main() -> AnyResult<()> { - tracing_subscriber::fmt() - .with_env_filter( - tracing_subscriber::EnvFilter::try_from_default_env() - .unwrap_or_else(|_| tracing_subscriber::EnvFilter::new("info,s3fs=debug")), - ) - .with_target(false) - .compact() - .init(); - - let cli = Cli::parse(); - - // ---- Build the S3-backed Fs. - let backend_cfg = AwsS3BackendConfig { - bucket: cli.bucket.clone(), - region: cli.region.clone(), - endpoint: cli.endpoint.clone(), - access_key_id: cli.access_key_id.clone(), - secret_access_key: cli.secret_access_key.clone(), - session_token: cli.session_token.clone(), - force_path_style: cli.force_path_style, - request_timeout: Duration::from_secs(30), - }; - let backend = if cli.skip_bucket_probe { - AwsS3Backend::connect_unchecked(backend_cfg).await? - } else { - AwsS3Backend::connect(backend_cfg).await? - }; - let fs_config = s3fs_core::Config::builder() - .bucket_prefix(cli.bucket_prefix.clone()) - .mount_path(cli.mount_path.clone()) - .build(); - let fs: Arc = Fs::new(Arc::new(backend) as Arc, Arc::new(fs_config)); - - tracing::info!( - bucket = %cli.bucket, - prefix = %cli.bucket_prefix, - mount = %cli.mount_path, - component = %cli.component.display(), - "loaded S3 backend, instantiating component" - ); - - // ---- Build the wasmtime engine + linker. - // wasmtime 44 enables async at the engine level by default when the - // `async` feature is on; `Config::async_support` is now a no-op. - let config = Config::new(); - let engine = Engine::new(&config)?; - - let mut linker: Linker = Linker::new(&engine); - add_wasi_minus_filesystem(&mut linker)?; - s3fs_wasmtime::add_to_linker(&mut linker).map_err(|e| anyhow::anyhow!(e.to_string()))?; - - let bytes = std::fs::read(&cli.component).context("reading component bytes")?; - let component = Component::new(&engine, &bytes) - .map_err(|e| anyhow::anyhow!(e.to_string())) - .with_context(|| format!("compiling component {}", cli.component.display()))?; - - // ---- Build the store with our State. - let mut wasi_builder = WasiCtxBuilder::new(); - wasi_builder.inherit_stdio(); - wasi_builder.envs(&std::env::vars().collect::>()); - wasi_builder.args(&cli.guest_args); - let state = State { - wasi: wasi_builder.build(), - table: ResourceTable::new(), - fs: fs.clone(), - }; - let mut store = Store::new(&engine, state); - - // ---- Instantiate and call the wasi:cli/run.run() export. - let command = - wasmtime_wasi::p2::bindings::Command::instantiate_async(&mut store, &component, &linker) - .await - .map_err(|e| anyhow::anyhow!(e.to_string())) - .context("instantiating wasi:cli/command component")?; - let run_result = command - .wasi_cli_run() - .call_run(&mut store) - .await - .map_err(|e| anyhow::anyhow!(e.to_string())) - .context("calling wasi:cli/run.run()")?; - - match run_result { - Ok(()) => { - tracing::info!("guest exited successfully"); - Ok(()) - } - Err(()) => { - anyhow::bail!("guest signalled failure (returned Err from wasi:cli/run.run())"); - } - } -} diff --git a/crates/s3fs-wasmtime/Cargo.toml b/crates/s3fs-wasmtime/Cargo.toml deleted file mode 100644 index 2212d91..0000000 --- a/crates/s3fs-wasmtime/Cargo.toml +++ /dev/null @@ -1,19 +0,0 @@ -[package] -name = "s3fs-wasmtime" -version.workspace = true -edition.workspace = true -rust-version.workspace = true -license.workspace = true -repository.workspace = true -description = "Wasmtime host bindings that expose `s3fs-core` as wasi:filesystem@0.2.x" - -[dependencies] -s3fs-core = { path = "../s3fs-core" } -wasmtime = { workspace = true } -wasmtime-wasi = { workspace = true } -wasmtime-wasi-io = "=44.0.0" -async-trait = { workspace = true } -anyhow = { workspace = true } -bytes = { workspace = true } -tokio = { workspace = true, features = ["sync", "rt", "macros"] } -tracing = { workspace = true } diff --git a/crates/s3fs-wasmtime/src/descriptors.rs b/crates/s3fs-wasmtime/src/descriptors.rs deleted file mode 100644 index 32c9f98..0000000 --- a/crates/s3fs-wasmtime/src/descriptors.rs +++ /dev/null @@ -1,67 +0,0 @@ -//! Resource types stored in the wasmtime `ResourceTable`. - -use std::sync::Arc; - -use s3fs_core::{FileHandle, Inode}; - -/// Wasmtime resource handle for `wasi:filesystem/types/descriptor`. -/// -/// A descriptor is one of two flavours: -/// - **File descriptor**: holds an `Arc` plus its parent inode -/// (so paths used in `_at` operations resolve relative to the right place). -/// - **Directory descriptor**: holds the directory inode directly. Reads / -/// writes against a dir-descriptor return `is-directory`. -/// -/// Both flavours can be used as the `base` for `*-at` operations; for a file -/// descriptor we resolve relative to its parent. -#[derive(Debug)] -pub enum Descriptor { - File { handle: Arc }, - Dir { inode: Arc }, -} - -impl Descriptor { - /// The inode this descriptor opens or is rooted under. For files, the - /// file's own inode; for dirs, the directory inode itself. - pub fn inode(&self) -> &Arc { - match self { - Descriptor::File { handle } => &handle.inode, - Descriptor::Dir { inode } => inode, - } - } - - /// The directory inode to use as the base for `*-at` path resolution. - /// For a file descriptor, that's the file's parent (or the file itself - /// if it has no parent — which would be an open of the root, an - /// invalid case for files). - pub fn at_base(&self) -> Arc { - match self { - Descriptor::Dir { inode } => inode.clone(), - Descriptor::File { handle } => { - // For an open file the parent should always exist; fall back - // to the file inode itself in the degenerate case. - handle - .inode - .parent - .as_ref() - .and_then(|w| w.upgrade()) - .unwrap_or_else(|| handle.inode.clone()) - } - } - } -} - -/// Wasmtime resource handle for `wasi:filesystem/types/directory-entry-stream`. -/// -/// Holds a snapshot of directory entries plus an iteration cursor. -#[derive(Debug)] -pub struct DirectoryEntryStream { - pub entries: Vec, - pub cursor: usize, -} - -impl DirectoryEntryStream { - pub fn new(entries: Vec) -> Self { - Self { entries, cursor: 0 } - } -} diff --git a/crates/s3fs-wasmtime/src/host_preopens.rs b/crates/s3fs-wasmtime/src/host_preopens.rs deleted file mode 100644 index ab83aa6..0000000 --- a/crates/s3fs-wasmtime/src/host_preopens.rs +++ /dev/null @@ -1,17 +0,0 @@ -//! `wasi:filesystem/preopens::Host` — exposes the single root descriptor. - -use wasmtime::component::Resource; -use wasmtime::Result; - -use crate::bindings::wasi::filesystem::preopens::Host; -use crate::bindings::wasi::filesystem::types::Descriptor as WitDescriptor; -use crate::descriptors::Descriptor; -use crate::view::S3FsCtxView; - -impl Host for S3FsCtxView<'_> { - async fn get_directories(&mut self) -> Result, String)>> { - let root = self.fs.root(); - let descriptor = self.table.push(Descriptor::Dir { inode: root })?; - Ok(vec![(descriptor, self.fs.config.mount_path.clone())]) - } -} diff --git a/crates/s3fs-wasmtime/src/lib.rs b/crates/s3fs-wasmtime/src/lib.rs deleted file mode 100644 index fef4cd2..0000000 --- a/crates/s3fs-wasmtime/src/lib.rs +++ /dev/null @@ -1,46 +0,0 @@ -//! `s3fs-wasmtime` — Wasmtime host bindings exposing `s3fs-core`'s `Fs` as -//! the `wasi:filesystem@0.2.x` interface. -//! -//! Designed to plug into a `wasmtime::component::Linker` alongside the rest -//! of `wasmtime-wasi` (which provides `wasi:io`, `wasi:cli`, clocks, random, -//! …). Only `wasi:filesystem/{types,preopens}` is overridden by this crate — -//! everything else stays on `wasmtime-wasi`'s implementations. -//! -//! Streams (`read-via-stream` / `write-via-stream`) are **not** yet wired up; -//! they currently return `error-code::unsupported`. The synchronous -//! `descriptor.read` / `descriptor.write` paths work end-to-end. - -pub mod bindings; -pub mod descriptors; -pub mod error_map; -pub mod host_filesystem; -pub mod host_preopens; -pub mod streams; -pub mod view; - -pub use descriptors::{Descriptor, DirectoryEntryStream}; -pub use view::{S3FsCtxView, S3WasiView}; - -use wasmtime::component::{HasData, Linker}; -use wasmtime::Result; - -/// `HasData` marker so bindgen knows the trait impls live on -/// [`S3FsCtxView<'_>`]. -pub struct HasS3Fs; -impl HasData for HasS3Fs { - type Data<'a> = S3FsCtxView<'a>; -} - -/// Add the `wasi:filesystem/{types,preopens}` interfaces to `linker`, backed -/// by an `s3fs-core::Fs` retrieved from the store via the [`S3WasiView`] trait. -/// -/// The caller is responsible for adding the rest of WASI (i/o, clocks, cli, -/// …) before or after this call. See `s3fs-runner` for a worked example. -pub fn add_to_linker(linker: &mut Linker) -> Result<()> { - fn getter(t: &mut T) -> S3FsCtxView<'_> { - t.s3fs_view() - } - bindings::wasi::filesystem::types::add_to_linker::(linker, getter::)?; - bindings::wasi::filesystem::preopens::add_to_linker::(linker, getter::)?; - Ok(()) -} diff --git a/deploy/Dockerfile b/deploy/Dockerfile new file mode 100644 index 0000000..c2c28ce --- /dev/null +++ b/deploy/Dockerfile @@ -0,0 +1,115 @@ +# Enclave image for `nitro-cli build-enclave`. +# +# The guest component is NOT in this image. The runtime fetches it at boot from +# `S3FS_GUEST_OBJECT`, a key in the roots bucket, extends PCR16 with its hash +# and locks that register before it asks KMS for anything. This image — PCR0 — +# covers the runtime and *where* the guest comes from; PCR16 covers *what* +# arrived. A key policy pinning both still attests to exactly which guest code +# will read the data, not merely to the runtime that loads it. +# +# Changing the guest therefore means uploading it and replacing PCR16 in the +# key policy, not rebuilding this image. The approval was always the feature; +# the rebuild never was. +# +# docker build -f deploy/Dockerfile -t s3fs-enclave . +# nitro-cli build-enclave --docker-uri s3fs-enclave:latest --output-file s3fs.eif +# nitro-cli run-enclave --eif-path s3fs.eif --cpu-count 2 --memory 2048 +# +# `build-enclave` prints PCR0/PCR1/PCR2; PCR0 is what the key policy pins for +# the runtime. The guest is built, measured and uploaded on its own: +# +# (cd examples/guest-http && cargo build --release --target wasm32-wasip2) +# GUEST=examples/guest-http/target/wasm32-wasip2/release/guest-http.wasm +# nitro-attest --measure "$GUEST" # the PCR16 the key policy pins +# aws s3 cp "$GUEST" s3:///guest/guest.wasm + +# ---- build ------------------------------------------------------------------ +FROM rust:1-bookworm AS build + +# aws-lc-rs builds a bundled C library. +RUN apt-get update && apt-get install -y --no-install-recommends \ + cmake clang \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /src +COPY . . + +RUN cargo build --release -p enclave-runtime + +# ---- the tap forwarder ------------------------------------------------------ +# An enclave has no NIC, so nothing that speaks TCP works until this turns the +# vsock channel into an interface. It ships *inside* the image rather than +# being fetched at boot, so PCR0 covers it: a forwarder streamed in later would +# be code the attestation says nothing about, sitting on every packet. +# +# Pinned by digest. A floating tag would change PCR0 without anything in this +# repository changing, which would look like the image had been tampered with. +FROM golang:1.23-bookworm AS gvforwarder +ARG GVISOR_TAP_VSOCK_VERSION=v0.8.6 +RUN git clone --depth 1 --branch ${GVISOR_TAP_VSOCK_VERSION} \ + https://github.com/containers/gvisor-tap-vsock /src \ + && cd /src \ + && CGO_ENABLED=0 go build -o /gvforwarder ./cmd/gvforwarder + +# ---- runtime ---------------------------------------------------------------- +FROM debian:bookworm-slim + +RUN apt-get update && apt-get install -y --no-install-recommends \ + ca-certificates \ + && rm -rf /var/lib/apt/lists/* + +COPY --from=build /src/target/release/enclave-runtime /usr/local/bin/enclave-runtime +COPY --from=gvforwarder /gvforwarder /usr/local/bin/gvforwarder + +# Deployment configuration. Every setting also has a flag, but inside an +# enclave there is no shell to pass one, so these are what actually apply. +# +# Where the guest comes from, as a key in S3FS_ROOTS_BUCKET. The location is +# measured here, by PCR0; what is found there is measured into PCR16 at boot, +# so the object does not have to be trusted — a substituted one gets no key. +ENV S3FS_GUEST_OBJECT=guest/guest.wasm +# Opt in for a guest implementing enclave:tasks/background@0.1.0, with WebAuthn +# configured. See docs/BACKGROUND_TASKS.md for standing authorization semantics. +# ENV S3FS_BACKGROUND_TASKS=true +# ENV S3FS_BACKGROUND_CONCURRENCY=1 +# Read the Nitro PTP hardware clock and refuse to start without it. The +# alternative is running on a clock the parent instance sets, which is the +# party this whole design excludes. +ENV S3FS_CLOCK_SOURCE=ptp +# Take entropy from the Nitro Security Module and refuse to start without it. +# The alternative is a kernel pool that is NSM-seeded in an enclave but says +# nothing about it, and the guest builds keys from these bytes. +ENV S3FS_RANDOM_SOURCE=nsm +# Bring up the tap device over vsock before anything opens a socket. The parent +# instance must be running `gvproxy --listen vsock://:1024` — see +# deploy/parent/run-parent.sh. Without it the runtime waits, then says so. +ENV S3FS_NETWORK=gvproxy +# Serve the guest over HTTPS, terminated here. A certificate generated inside +# the enclave, with its hash signed into the attestation document, is what lets +# a client prove its session ends in *this* code rather than in the parent. +ENV S3FS_HTTP_LISTEN=0.0.0.0:443 +# ACME, and only ACME. A self-signed certificate is one an operator can mint +# too, so it cannot tell this enclave apart from something impersonating it — +# the runtime refuses the setting rather than generating one. +ENV S3FS_TLS=acme +ENV S3FS_TLS_DOMAINS=enclave.example.com +# ENV S3FS_BUCKET=prod-data +# ENV S3FS_ROOTS_BUCKET=prod-roots # Object Lock COMPLIANCE lives here +# ENV AWS_REGION=eu-west-2 +# ENV S3FS_MIN_ROOT_SEQ=0 # freshness floor from outside the store +# +# The master secret is minted by KMS inside the enclave and released only +# against an attestation whose PCR0 and PCR16 match the key policy. A wrong +# image or a wrong guest does not get a refused mount; it gets no key at all. +ENV S3FS_MASTER_KEY_SOURCE=kms +# ENV S3FS_KMS_KEY_ID=arn:aws:kms:eu-west-2:...:key/... +# ENV S3FS_MASTER_KEY_PARAMETER=/enclave-runtime/prod//master-key +# ENV S3FS_ENVIRONMENT=production # part of the encryption context +# +# S3FS_MASTER_KEY is deliberately NOT set, and the runtime *refuses to start* +# if it is set alongside the line above. A key baked into the image is readable +# by anyone who can pull the image, and one passed by the parent is readable by +# the party the enclave exists to exclude — so the combination is an error +# rather than a precedence rule. + +ENTRYPOINT ["/usr/local/bin/enclave-runtime"] diff --git a/deploy/ami/enclave-parent.pkr.hcl b/deploy/ami/enclave-parent.pkr.hcl new file mode 100644 index 0000000..d504ba4 --- /dev/null +++ b/deploy/ami/enclave-parent.pkr.hcl @@ -0,0 +1,202 @@ +# The AMI for the parent instance that hosts the enclave. +# +# Built with Packer over Amazon Linux 2023, not with Nix, and the split is not +# arbitrary: the parent is the party the enclave *excludes*. PCR0 covers what +# runs inside the enclave and says nothing about its host, so reproducing this +# machine buys no security property. AWS ships and supports the Nitro Enclaves +# CLI and allocator on AL2023, which is worth more here than a bit-identical +# image would be. +# +# The EIF is passed in rather than fetched, because *that* artifact is +# attested. It comes from `nix build .#eif`, which is reproducible, and baking +# it means the running enclave's PCR0 traces back to a build anyone can repeat. +# +# nix build .#eif +# nix build .#gvproxy +# packer init deploy/ami +# packer build \ +# -var eif=$(readlink -f result)/s3fs.eif \ +# -var pcr_json=$(readlink -f result)/pcr.json \ +# -var gvproxy=$(readlink -f result-1)/bin/gvproxy \ +# deploy/ami + +packer { + required_plugins { + amazon = { + version = ">= 1.3.0" + source = "github.com/hashicorp/amazon" + } + } +} + +variable "region" { + type = string + default = "eu-west-2" +} + +variable "instance_type" { + type = string + # Building needs no enclave support — only running does — so this is chosen + # for build speed rather than from the enclave-capable list. + default = "c6i.large" + description = "Instance type used to build the image" +} + +variable "eif" { + type = string + description = "Path to the EIF from `nix build .#eif`" +} + +variable "pcr_json" { + type = string + description = "Path to pcr.json from the same build, baked alongside the EIF" +} + +variable "gvproxy" { + type = string + description = "Path to the static gvproxy from `nix build .#gvproxy`" +} + +variable "enclave_cpu_count" { + type = number + default = 2 + description = "vCPUs the allocator reserves for enclaves" +} + +variable "enclave_memory_mib" { + type = number + # The EIF is ~140 MiB (a Nix closure, not a stripped musl binary) and the + # enclave needs room for the image plus the runtime's own heap. The 512 MiB + # default would fail at `run-enclave` with a message about memory that does + # not mention the image size. + default = 3072 + description = "Memory the allocator reserves for enclaves, in MiB" +} + +variable "ami_name_prefix" { + type = string + default = "s3fs-enclave-parent" +} + +locals { + timestamp = regex_replace(timestamp(), "[- TZ:]", "") +} + +source "amazon-ebs" "parent" { + region = var.region + instance_type = var.instance_type + ssh_username = "ec2-user" + + ami_name = "${var.ami_name_prefix}-${local.timestamp}" + ami_description = "Nitro Enclaves parent: nitro-cli, gvproxy, and a pinned s3fs EIF" + + source_ami_filter { + filters = { + name = "al2023-ami-2023.*-kernel-6.*-x86_64" + virtualization-type = "hvm" + root-device-type = "ebs" + } + owners = ["amazon"] + most_recent = true + } + + # The EIF alone is ~140 MiB; the default 8 GiB root leaves ample room but is + # stated so a larger image does not silently fill the disk. + launch_block_device_mappings { + device_name = "/dev/xvda" + volume_size = 16 + volume_type = "gp3" + delete_on_termination = true + } + + tags = { + Name = "${var.ami_name_prefix}-${local.timestamp}" + Component = "nitro-enclave-parent" + } +} + +build { + sources = ["source.amazon-ebs.parent"] + + # ---- the Nitro Enclaves CLI ------------------------------------------- + # + # Deliberately no Docker. It is needed only by `nitro-cli build-enclave`, + # and the image is built by Nix — so the untrusted host carries one fewer + # daemon, and there is no path by which an image could be rebuilt here into + # something with a different PCR0. + provisioner "shell" { + inline = [ + "set -euxo pipefail", + "sudo dnf -y update", + "sudo dnf -y install aws-nitro-enclaves-cli jq", + "sudo usermod -aG ne ec2-user", + "nitro-cli --version", + ] + } + + # ---- the artifacts ------------------------------------------------------ + provisioner "shell" { + inline = ["sudo mkdir -p /opt/enclave && sudo chown ec2-user /opt/enclave"] + } + + provisioner "file" { + source = var.eif + destination = "/opt/enclave/s3fs.eif" + } + + # The measurements the image claims, kept next to it. This is what lets an + # operator compare a running enclave's attested PCR0 against a rebuild — + # without it, a reproducible build proves nothing to anybody on the box. + provisioner "file" { + source = var.pcr_json + destination = "/opt/enclave/pcr.json" + } + + # The same pin as the gvforwarder inside the EIF: one nixpkgs package emits + # both ends of the vsock, so they cannot drift apart. Static, because there + # is no Nix store on this machine to link against. + provisioner "file" { + source = var.gvproxy + destination = "/opt/enclave/gvproxy" + } + + # ---- allocator ---------------------------------------------------------- + provisioner "shell" { + inline = [ + "set -euxo pipefail", + "sudo chmod +x /opt/enclave/gvproxy", + "sudo install -m 0755 /opt/enclave/gvproxy /usr/local/bin/gvproxy", + "sudo tee /etc/nitro_enclaves/allocator.yaml >/dev/null <<'YAML'", + "---", + "memory_mib: ${var.enclave_memory_mib}", + "cpu_count: ${var.enclave_cpu_count}", + "YAML", + "sudo systemctl enable nitro-enclaves-allocator.service", + ] + } + + # ---- systemd ------------------------------------------------------------ + provisioner "file" { + source = "${path.root}/units/" + destination = "/tmp/units" + } + + provisioner "shell" { + inline = [ + "set -euxo pipefail", + "sudo install -m 0644 /tmp/units/gvproxy.service /etc/systemd/system/", + "sudo install -m 0644 /tmp/units/enclave.service /etc/systemd/system/", + "sudo install -m 0755 /tmp/units/enclave-start.sh /usr/local/bin/enclave-start", + "sudo systemctl enable gvproxy.service enclave.service", + # Fail the build rather than the boot: a unit that will not even parse is + # discovered here, not at 3am on a machine with no shell access. + "sudo systemd-analyze verify /etc/systemd/system/gvproxy.service", + "sudo systemd-analyze verify /etc/systemd/system/enclave.service", + ] + } + + post-processor "manifest" { + output = "deploy/ami/manifest.json" + strip_path = true + } +} diff --git a/deploy/ami/units/enclave-start.sh b/deploy/ami/units/enclave-start.sh new file mode 100644 index 0000000..3cf7648 --- /dev/null +++ b/deploy/ami/units/enclave-start.sh @@ -0,0 +1,65 @@ +#!/usr/bin/env bash +# Start the enclave and open the door to it. +# +# Two steps that have to happen in this order and are easy to get wrong: +# +# 1. `nitro-cli run-enclave` boots the EIF. It returns as soon as the enclave +# is running, not when the application inside it is listening. +# 2. gvproxy is told to forward :443 into the enclave. This is a call to its +# API socket rather than a flag, which means the enclave can be restarted +# without restarting the network under it. +# +# The forwarding is deliberately *not* done before the enclave exists: gvproxy +# would accept the request and then refuse every connection, which looks like a +# firewall problem rather than a missing enclave. +set -euo pipefail + +EIF="${EIF:-/opt/enclave/s3fs.eif}" +CPU_COUNT="${CPU_COUNT:-2}" +MEMORY_MIB="${MEMORY_MIB:-3072}" +ENCLAVE_IP="${ENCLAVE_IP:-192.168.127.2}" +FORWARD_PORTS="${FORWARD_PORTS:-443}" +API_SOCKET="${API_SOCKET:-/run/gvproxy/network.sock}" +ENCLAVE_CID="${ENCLAVE_CID:-16}" + +log() { echo "enclave-start: $*"; } + +[[ -f "$EIF" ]] || { log "no image at $EIF"; exit 1; } + +# What this machine is about to run, so the journal records the measurement +# rather than only the fact that something started. An operator comparing a +# client's attested PCR0 against a rebuild starts here. +if [[ -f /opt/enclave/pcr.json ]]; then + log "PCR0 $(jq -r .PCR0 /opt/enclave/pcr.json)" +fi + +# A previous enclave surviving a restart would hold the allocator's memory and +# the next `run-enclave` would fail for lack of it. +if nitro-cli describe-enclaves | jq -e '.[0]' >/dev/null 2>&1; then + log "terminating a previously running enclave" + nitro-cli terminate-enclave --all >/dev/null +fi + +log "starting enclave from $EIF (${CPU_COUNT} vCPU, ${MEMORY_MIB} MiB)" +nitro-cli run-enclave \ + --eif-path "$EIF" \ + --cpu-count "$CPU_COUNT" \ + --memory "$MEMORY_MIB" \ + --enclave-cid "$ENCLAVE_CID" + +for _ in $(seq 1 60); do + [[ -S "$API_SOCKET" ]] && break + sleep 1 +done +[[ -S "$API_SOCKET" ]] || { log "gvproxy's API socket never appeared at $API_SOCKET"; exit 1; } + +for port in $FORWARD_PORTS; do + log "forwarding :$port to ${ENCLAVE_IP}:$port" + curl -sf --unix-socket "$API_SOCKET" \ + http://localhost/services/forwarder/expose \ + -X POST -H 'Content-Type: application/json' \ + -d "{\"local\":\":${port}\",\"remote\":\"${ENCLAVE_IP}:${port}\"}" \ + || { log "gvproxy refused to forward :$port"; exit 1; } +done + +log "enclave is running and :$FORWARD_PORTS is forwarded" diff --git a/deploy/ami/units/enclave.service b/deploy/ami/units/enclave.service new file mode 100644 index 0000000..4d9884e --- /dev/null +++ b/deploy/ami/units/enclave.service @@ -0,0 +1,31 @@ +[Unit] +Description=s3fs Nitro Enclave +# Ordered after both, because the enclave needs memory the allocator reserves +# and a network gvproxy provides. Started before either, it fails in ways that +# name neither. +After=nitro-enclaves-allocator.service gvproxy.service +Requires=nitro-enclaves-allocator.service gvproxy.service + +[Service] +# `run-enclave` returns once the enclave is running rather than staying in the +# foreground, so this is oneshot with RemainAfterExit — the unit's state then +# tracks "an enclave was started", which is what it can honestly claim. The +# enclave's own liveness is a question for its attestation endpoint, not +# systemd. +Type=oneshot +RemainAfterExit=yes +ExecStart=/usr/local/bin/enclave-start +ExecStop=/usr/bin/nitro-cli terminate-enclave --all + +Environment=EIF=/opt/enclave/s3fs.eif +Environment=CPU_COUNT=2 +Environment=MEMORY_MIB=3072 +Environment=FORWARD_PORTS=443 + +# The enclave takes a while to load a ~140 MiB image and mount its filesystem. +TimeoutStartSec=300 +Restart=on-failure +RestartSec=10 + +[Install] +WantedBy=multi-user.target diff --git a/deploy/ami/units/gvproxy.service b/deploy/ami/units/gvproxy.service new file mode 100644 index 0000000..26856f3 --- /dev/null +++ b/deploy/ami/units/gvproxy.service @@ -0,0 +1,35 @@ +# The enclave's only route to anything. +# +# An enclave has no NIC. Every packet it sends leaves over vsock and arrives +# here, so nothing inside works until this is running — not S3, not DNS, not an +# inbound request. An enclave that boots and then answers nothing is almost +# always this unit having failed. +[Unit] +Description=gvproxy, the enclave's network +Before=enclave.service +After=network-online.target +Wants=network-online.target + +[Service] +Type=exec +ExecStart=/usr/local/bin/gvproxy \ + --listen vsock://:1024 \ + --listen unix:///run/gvproxy/network.sock +RuntimeDirectory=gvproxy +Restart=always +RestartSec=2 + +# It terminates ethernet frames from an untrusted enclave and speaks to the +# network on its behalf, so it gets no more of the host than that needs. +DynamicUser=no +User=root +NoNewPrivileges=yes +ProtectSystem=strict +ProtectHome=yes +PrivateTmp=yes +ProtectKernelModules=yes +ProtectControlGroups=yes +RestrictSUIDSGID=yes + +[Install] +WantedBy=multi-user.target diff --git a/deploy/nix/README.md b/deploy/nix/README.md new file mode 100644 index 0000000..2512754 --- /dev/null +++ b/deploy/nix/README.md @@ -0,0 +1,80 @@ +# Building the enclave image + +```console +$ nix build .#eif # → result/s3fs.eif, result/pcr.json +$ jq -r .PCR0 result/pcr.json +42dfa2f7f828fd1dbb955a8d8c00537c4545d56467e850254f72bd80627c35fa… +``` + +## Why Nix, only here + +PCR0 is a digest of the enclave image. A KMS key policy pins it and a client +checks it, and the whole value of that number is that somebody else can rebuild +the image and get the same one. The Dockerfile this replaced ran `apt-get +update` over `debian:bookworm-slim`, so it produced a different PCR0 every week +and attested to nothing anyone could reproduce. + +The parent instance is built with Packer instead, and that is not +inconsistency. The parent is the party the enclave *excludes*: PCR0 covers what +runs inside the enclave and says nothing about its host, so reproducing the +host buys no security property at all. + +## Outputs + +| | | +|---|---| +| `.#eif` | the production image, with `pcr.json` | +| `.#eif-qemu` | the same code configured for the emulator — different config, therefore a different PCR0 | +| `.#eif-selftest` | the tiny NSM entropy check | +| `.#enclave-runtime`, `.#guest-http`, `.#nitro-attest` | the binaries | +| `.#gvproxy` | both ends of the vsock, built static | + +## Reproducibility, and what it cost + +`nix build .#eif --rebuild` rebuilds and compares. Getting that to pass took +four things, none of which was obvious from a failure message: + +- **`cpio --reproducible`.** The `newc` header records each file's inode and + device numbers, which are whatever the filesystem handed out. Without this + even the *bootstrap* ramdisk — two files that never change — came out + different on every build, and PCR1 with it. +- **`--owner=0:0` and a fixed mtime.** Store paths already carry epoch+1; the + directories created during assembly do not. +- **Sorted members.** cpio records the order it is given, and `find` does not + promise one. `LC_ALL=C sort` also keeps parents ahead of their children. +- **`faketime`.** `eif_build` stamps wall-clock `BuildTime` into the image's + metadata. No PCR covers metadata, so the *measurements* were reproducible + without this — but the file differed by 15 bytes, which is enough to make + anyone comparing artifacts by hash think something was wrong. + +A pinned lock file for `eif_build` is in this directory for a related reason: +upstream gitignores theirs, and letting the dependency set of the tool that +*computes PCR0* float would mean the measurement came from something slightly +different each time. + +## What is still taken on trust + +The kernel, `init` and `nsm.ko` are AWS's prebuilt blobs, pinned by hash. Every +build therefore uses identical bytes, but their provenance is not verifiable +from here — they are the one opaque input to PCR0. AWS publishes +[`aws-nitro-enclaves-sdk-bootstrap`](https://github.com/aws/aws-nitro-enclaves-sdk-bootstrap), +which builds them from source *with nixpkgs*; adopting it would close the gap +at the cost of a kernel compile. + +## Running Nix without root + +The flake is ordinary and works with any Nix. On a machine where you cannot +install the daemon, [`nix-portable`](https://github.com/DavHau/nix-portable) +gives a working `nix` in userspace: + +```console +$ curl -L -o ~/.local/bin/nix-portable \ + https://github.com/DavHau/nix-portable/releases/latest/download/nix-portable-x86_64 +$ chmod +x ~/.local/bin/nix-portable && ln -s ~/.local/bin/{nix-portable,nix} +``` + +Two consequences worth knowing. `nix run` and `nix shell` need a mount +namespace it cannot always get, so use `nix build` and run the result. And the +store lives at `~/.nix-portable/nix/store`, so `result/` is a symlink that only +resolves inside its namespace — the harness scripts translate the path, and +that branch is dead on a normal install. diff --git a/deploy/nix/deployment.nix b/deploy/nix/deployment.nix new file mode 100644 index 0000000..b2fb341 --- /dev/null +++ b/deploy/nix/deployment.nix @@ -0,0 +1,114 @@ +# Which store this enclave image belongs to. +# +# These are baked into the image, so PCR0 covers them — and that is the point. +# A state-origin receipt proves an enclave of a known image created this +# filesystem, but only if the host cannot point that image at a *different* +# bucket: an empty one has no receipt, so it would take the genesis path and +# serve a fresh empty filesystem with every check passing. +# +# The consequence is real and worth knowing before you hit it: **changing any +# value here changes PCR0**. Two deployments are two images, a KMS key policy +# pinned to one will not release to the other, and clients pin different +# measurements. The enclave's identity includes what it operates on. +# +# `S3FS_ID` is not a secret. It is the HKDF salt, so two filesystems under one +# master secret stay independent, and it must be supplied rather than read from +# the store: the keys that verify a root record derive from it, so taking it +# from the store would mean trusting the store to say which key checks its own +# signature. +{ + dataBucket = "CHANGE-ME-data"; + rootsBucket = "CHANGE-ME-roots"; + bucketPrefix = ""; + + # 32 hex characters. + fsId = "00000000000000000000000000000000"; + + region = "eu-west-2"; + + # Domains for the serving certificate. Required, and not only for browsers: + # the runtime obtains its certificate over ACME and nothing else, so an empty + # list means no certificate and no HTTPS at all. A platform authenticator + # also refuses to attest against a certificate a browser does not trust, so a + # deployment that authenticates needs a real domain here. + tlsDomains = [ ]; + + # The domain passkeys are scoped to, and the origin assertions must claim. + # + # Must be one of `tlsDomains`: a passkey is bound to a domain, and an + # assertion carries the origin the page was served from, compared exactly. + # Baked into the image, so PCR0 records which relying party an enclave will + # accept assertions for — a client can verify that before trusting it with a + # key. + rpId = "CHANGE-ME.example.com"; + + # Origins other than `https://${rpId}` that assertions may claim — native + # apps, which never claim the web origin. An Android app claims + # "android:apk-key-hash:", the unpadded base64url SHA-256 of the + # certificate it is signed with; for a Play release that is the app signing + # key, not the upload key. `keytool` prints the colon-separated hex form of + # the same digest, which must be converted — the runtime refuses it at boot. + # + # Android lets an app claim this only if + # `https://${rpId}/.well-known/assetlinks.json` lists it, so that file must + # be served too. Measured by PCR0 like `rpId`: adding an app is a new image. + webauthnAllowedOrigins = [ ]; + + # Where guest stdout and stderr go, on top of the enclave console. + # + # Both must already exist — `deploy/tofu` creates them, and the enclave holds + # `logs:PutLogEvents` and nothing more, so it cannot create them itself. They + # are baked into the image and therefore measured by PCR0, which is why a + # client can tell from an attestation where an enclave ships guest output. + # + # The group must match `aws_cloudwatch_log_group.guest` in `deploy/tofu`, + # which names it "/${name_prefix}-${environment}/guest". + guestLogGroup = "/CHANGE-ME-production/guest"; + guestLogStream = "guest"; + + # Where the runtime fetches its guest: a key in `rootsBucket`, used verbatim + # (`bucketPrefix` is not applied). + # + # The key is measured by PCR0 like everything else here. The object behind it + # is measured by the enclave into PCR16 at boot, before it asks KMS for a key, + # so it need not be trusted: changing the guest is an upload and a key-policy + # edit, not a new image. Upload `guest-release/guest.wasm` here. + guestObject = "guest/guest.wasm"; + + # Opt in only for a guest implementing enclave:tasks/background@0.1.0. + # Each task is authorized by an authenticated tenant interaction. Queued + # work survives restarts; only this active enclave may own its scheduler. + backgroundTasks = false; + backgroundConcurrency = 1; + # How long one background task may run. A task that drives a round with an + # outside service waits on that service's schedule, which is minutes. `null` + # keeps the runtime's default (30 seconds) and leaves PCR0 as it was. + backgroundTimeoutSecs = null; + + # Origins guests may send requests to, as "https://host[:port]". None by + # default, and then a guest has no outbound network at all. + # + # Compared exactly — scheme, host, port — over TLS verified against the web + # PKI. It is a channel out of the enclave carrying whatever the guest puts in + # it, so name only services the guest has to reach: for a wallet cosigner, + # its ASP. Measured by PCR0 like everything here: a client learns where the + # guest can send traffic from the attestation, and adding one is a new image. + guestEgressOrigins = [ ]; + + # Variables for the guest, as { NAME = "value"; }. The guest inherits the image + # environment minus anything under `AWS_` or `S3FS_`, so these reach it — and, + # being image environment, are measured by PCR0 like the rest. For a cosigner + # whose egress names its ASP, this is where it learns the address: { ASP_URL = + # "https://asp.example.com"; }. + guestEnv = { }; + + # Push notifications, off unless both are set. `fcmProjectId` is the Firebase + # project; the service account itself is read at boot from this SSM parameter + # rather than baked in, so it rotates without moving PCR0. + # + # What a parent that steals that credential gets is the ability to ring + # doorbells: a wake signal carries no content, and reading anything still + # needs a key KMS releases only against a matching PCR0 and PCR16. + fcmProjectId = ""; + fcmServiceAccountParameter = ""; +} diff --git a/deploy/nix/eif_build-Cargo.lock b/deploy/nix/eif_build-Cargo.lock new file mode 100644 index 0000000..3ede775 --- /dev/null +++ b/deploy/nix/eif_build-Cargo.lock @@ -0,0 +1,2925 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "aho-corasick" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" +dependencies = [ + "memchr", +] + +[[package]] +name = "android_system_properties" +version = "0.1.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" +dependencies = [ + "libc", +] + +[[package]] +name = "anstream" +version = "0.6.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "43d5b281e737544384e969a5ccad3f1cdd24b48086a0fc1b2a5262a26b8f4f4a" +dependencies = [ + "anstyle", + "anstyle-parse", + "anstyle-query", + "anstyle-wincon", + "colorchoice", + "is_terminal_polyfill", + "utf8parse", +] + +[[package]] +name = "anstyle" +version = "1.0.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "940b3a0ca603d1eade50a4846a2afffd5ef57a9feac2c0e2ec2e14f9ead76000" + +[[package]] +name = "anstyle-parse" +version = "0.2.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4e7644824f0aa2c7b9384579234ef10eb7efb6a0deb83f9630a49594dd9c15c2" +dependencies = [ + "utf8parse", +] + +[[package]] +name = "anstyle-query" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "40c48f72fd53cd289104fc64099abca73db4166ad86ea0b4341abe65af83dadc" +dependencies = [ + "windows-sys 0.61.2", +] + +[[package]] +name = "anstyle-wincon" +version = "3.0.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "291e6a250ff86cd4a820112fb8898808a366d8f9f58ce16d1f538353ad55747d" +dependencies = [ + "anstyle", + "once_cell_polyfill", + "windows-sys 0.61.2", +] + +[[package]] +name = "atomic-waker" +version = "1.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" + +[[package]] +name = "atty" +version = "0.2.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d9b39be18770d11421cdb1b9947a45dd3f37e93092cbf377614828a319d5fee8" +dependencies = [ + "hermit-abi", + "libc", + "winapi", +] + +[[package]] +name = "autocfg" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" + +[[package]] +name = "aws-config" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "48730d0b4c3d91c43d0d37168831d9fd0e065ad4a889a2ee9faf8d34c3d2804d" +dependencies = [ + "aws-credential-types", + "aws-runtime", + "aws-sdk-sso", + "aws-sdk-ssooidc", + "aws-sdk-sts", + "aws-smithy-async 1.3.0", + "aws-smithy-http 0.60.12", + "aws-smithy-json", + "aws-smithy-runtime 1.12.1", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "aws-types", + "bytes", + "fastrand", + "hex", + "http 0.2.12", + "hyper", + "ring", + "time", + "tokio", + "tracing", + "url", + "zeroize", +] + +[[package]] +name = "aws-credential-types" +version = "1.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e93964ffdaf57857f544be3666a5f57570bb699e934700f11b49708f61bb556e" +dependencies = [ + "aws-smithy-async 1.3.0", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "zeroize", +] + +[[package]] +name = "aws-nitro-enclaves-cose" +version = "0.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b8a94047bd9c3717c6ca3a145504c0e26b64a5e2d9eb9559b187748433fbc382" +dependencies = [ + "aws-sdk-kms", + "openssl", + "serde", + "serde_bytes", + "serde_cbor", + "serde_repr", + "serde_with", + "tokio", +] + +[[package]] +name = "aws-nitro-enclaves-image-format" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "be2355fd67972562e7402384bf20c7d913070e4f0d0d1d722c4a3cbc3c977d6c" +dependencies = [ + "aws-config", + "aws-nitro-enclaves-cose", + "aws-sdk-kms", + "aws-smithy-runtime 0.101.0", + "aws-types", + "byteorder", + "chrono", + "clap 3.2.25", + "crc", + "hex", + "num-derive", + "num-traits", + "openssl", + "regex", + "serde", + "serde_cbor", + "serde_json", + "sha2 0.9.9", + "tokio", +] + +[[package]] +name = "aws-nitro-enclaves-image-format" +version = "0.6.0" +dependencies = [ + "aws-config", + "aws-nitro-enclaves-cose", + "aws-sdk-kms", + "aws-smithy-runtime 1.12.1", + "aws-types", + "byteorder", + "chrono", + "crc", + "hex", + "litemap 0.7.4", + "num-derive", + "num-traits", + "openssl", + "regex", + "serde", + "serde_cbor", + "serde_json", + "sha2 0.10.9", + "tempfile", + "tokio", + "zerofrom", +] + +[[package]] +name = "aws-runtime" +version = "1.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c4ee6903f9d0197510eb6b44c4d86b493011d08b4992938f7b9be0333b6685aa" +dependencies = [ + "aws-credential-types", + "aws-sigv4", + "aws-smithy-async 1.3.0", + "aws-smithy-http 0.60.12", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "aws-types", + "bytes", + "fastrand", + "http 0.2.12", + "http-body 0.4.6", + "percent-encoding", + "pin-project-lite", + "tracing", + "uuid", +] + +[[package]] +name = "aws-sdk-kms" +version = "1.20.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5cfadd284d25d59c715bec5d1b6a20724eea3f85332a632f56056ce3fc6ea934" +dependencies = [ + "aws-credential-types", + "aws-runtime", + "aws-smithy-async 1.3.0", + "aws-smithy-http 0.60.12", + "aws-smithy-json", + "aws-smithy-runtime 1.12.1", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "aws-types", + "bytes", + "http 0.2.12", + "once_cell", + "regex-lite", + "tracing", +] + +[[package]] +name = "aws-sdk-sso" +version = "1.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b2be5ba83b077b67a6f7a1927eb6b212bf556e33bd74b5eaa5aa6e421910803a" +dependencies = [ + "aws-credential-types", + "aws-runtime", + "aws-smithy-async 1.3.0", + "aws-smithy-http 0.60.12", + "aws-smithy-json", + "aws-smithy-runtime 1.12.1", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "aws-types", + "bytes", + "http 0.2.12", + "once_cell", + "regex-lite", + "tracing", +] + +[[package]] +name = "aws-sdk-ssooidc" +version = "1.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "022ca669825f841aef17b12d4354ef2b8651e4664be49f2d9ea13e4062a80c9f" +dependencies = [ + "aws-credential-types", + "aws-runtime", + "aws-smithy-async 1.3.0", + "aws-smithy-http 0.60.12", + "aws-smithy-json", + "aws-smithy-runtime 1.12.1", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "aws-types", + "bytes", + "http 0.2.12", + "once_cell", + "regex-lite", + "tracing", +] + +[[package]] +name = "aws-sdk-sts" +version = "1.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e4a5f5cb007347c1ab34a6d56456301dfada921fc9e57d687ecb08baddd11ff" +dependencies = [ + "aws-credential-types", + "aws-runtime", + "aws-smithy-async 1.3.0", + "aws-smithy-http 0.60.12", + "aws-smithy-json", + "aws-smithy-query", + "aws-smithy-runtime 1.12.1", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "aws-smithy-xml", + "aws-types", + "http 0.2.12", + "once_cell", + "regex-lite", + "tracing", +] + +[[package]] +name = "aws-sigv4" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "723c2234ad7511ceef63eab016b7ba6ff7c55590fefb96fa8467af014a07309f" +dependencies = [ + "aws-credential-types", + "aws-smithy-http 0.64.0", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "bytes", + "form_urlencoded", + "hex", + "hmac", + "http 0.2.12", + "http 1.5.0", + "percent-encoding", + "sha2 0.11.0", + "time", + "tracing", +] + +[[package]] +name = "aws-smithy-async" +version = "0.101.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d787b7e07925b450bed90d9d29ac8e57006c9c2ac907151d175ac0e376bfee0e" +dependencies = [ + "futures-util", + "pin-project-lite", + "tokio", +] + +[[package]] +name = "aws-smithy-async" +version = "1.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f02e407fb3b54891734224b9ffac8a71fdd35f542500fa1af95754a6b2beb316" +dependencies = [ + "futures-util", + "pin-project-lite", + "tokio", +] + +[[package]] +name = "aws-smithy-http" +version = "0.59.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "96daaad925331c72449423574fdc72b54af780d5a23ace3c0a6ad0ccbf378715" +dependencies = [ + "aws-smithy-runtime-api 0.101.0", + "aws-smithy-types 0.101.0", + "bytes", + "bytes-utils", + "futures-core", + "http 0.2.12", + "http-body 0.4.6", + "once_cell", + "percent-encoding", + "pin-project-lite", + "pin-utils", + "tracing", +] + +[[package]] +name = "aws-smithy-http" +version = "0.60.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7809c27ad8da6a6a68c454e651d4962479e81472aa19ae99e59f9aba1f9713cc" +dependencies = [ + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "bytes", + "bytes-utils", + "futures-core", + "http 0.2.12", + "http-body 0.4.6", + "once_cell", + "percent-encoding", + "pin-project-lite", + "pin-utils", + "tracing", +] + +[[package]] +name = "aws-smithy-http" +version = "0.64.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "37843d9add67c3aff5856f409c6dc315d3cdff60f9c0cb5b670dab1e9920306d" +dependencies = [ + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "bytes", + "bytes-utils", + "futures-core", + "futures-util", + "http 1.5.0", + "http-body 1.1.0", + "http-body-util", + "percent-encoding", + "pin-project-lite", + "pin-utils", + "tracing", +] + +[[package]] +name = "aws-smithy-http-client" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "635d23afda0a6ab48d666c4d447c4873e8d1e83518a2be2093122397e50b838e" +dependencies = [ + "aws-smithy-async 1.3.0", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "h2 0.3.27", + "h2 0.4.15", + "http 0.2.12", + "http-body 0.4.6", + "hyper", + "hyper-rustls", + "pin-project-lite", + "rustls", + "rustls-native-certs", + "tokio", + "tracing", +] + +[[package]] +name = "aws-smithy-json" +version = "0.60.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4683df9469ef09468dad3473d129960119a0d3593617542b7d52086c8486f2d6" +dependencies = [ + "aws-smithy-types 1.6.1", +] + +[[package]] +name = "aws-smithy-observability" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e86338c869539a581bf161247762a6e87f92c5c075060057b5ed6d06632ed0c" +dependencies = [ + "aws-smithy-runtime-api 1.14.0", +] + +[[package]] +name = "aws-smithy-query" +version = "0.60.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1a56d79744fb3edb5d722ef79d86081e121d3b9422cb209eb03aea6aa4f21ebd" +dependencies = [ + "aws-smithy-types 1.6.1", + "urlencoding", +] + +[[package]] +name = "aws-smithy-runtime" +version = "0.101.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d28af854558601b4202a4273b9720aebe43d73e472143e6056f16e3bd90bc837" +dependencies = [ + "aws-smithy-async 0.101.0", + "aws-smithy-http 0.59.0", + "aws-smithy-runtime-api 0.101.0", + "aws-smithy-types 0.101.0", + "bytes", + "fastrand", + "http 0.2.12", + "http-body 0.4.6", + "once_cell", + "pin-project-lite", + "pin-utils", + "tokio", + "tracing", +] + +[[package]] +name = "aws-smithy-runtime" +version = "1.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "07505b34e8f4b3591a4fa69e9792b52289b95488dbbc68c3c0075b7bedb245e1" +dependencies = [ + "aws-smithy-async 1.3.0", + "aws-smithy-http 0.64.0", + "aws-smithy-http-client", + "aws-smithy-observability", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-schema", + "aws-smithy-types 1.6.1", + "bytes", + "fastrand", + "http 0.2.12", + "http 1.5.0", + "http-body 0.4.6", + "http-body 1.1.0", + "http-body-util", + "pin-project-lite", + "pin-utils", + "tokio", + "tracing", +] + +[[package]] +name = "aws-smithy-runtime-api" +version = "0.101.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e1c68e17e754b86da350b43add38294189121a880e9c3fb454f83ff7044f5257" +dependencies = [ + "aws-smithy-async 0.101.0", + "aws-smithy-types 0.101.0", + "bytes", + "http 0.2.12", + "pin-project-lite", + "tokio", + "tracing", +] + +[[package]] +name = "aws-smithy-runtime-api" +version = "1.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3b98f2e1fd67ec06618f9c291e5e495a468e60519e44c9c1979cd0521f3affdb" +dependencies = [ + "aws-smithy-async 1.3.0", + "aws-smithy-runtime-api-macros", + "aws-smithy-types 1.6.1", + "bytes", + "http 0.2.12", + "http 1.5.0", + "pin-project-lite", + "tokio", + "tracing", + "zeroize", +] + +[[package]] +name = "aws-smithy-runtime-api-macros" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "221eaa237ddf1ca79b60d1372aad77e47f9c0ea5b3ce5099da8c61d027dc77b3" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "aws-smithy-schema" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7d56e0a4e53127a632224e43633b0fe045fa9e1e3cfc68b9830f1115e103f910" +dependencies = [ + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "http 1.5.0", +] + +[[package]] +name = "aws-smithy-types" +version = "0.101.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d97b978d8a351ea5744206ecc643a1d3806628680e9f151b4d6b7a76fec1596f" +dependencies = [ + "base64-simd", + "bytes", + "bytes-utils", + "futures-core", + "http 0.2.12", + "http-body 0.4.6", + "itoa", + "num-integer", + "pin-project-lite", + "pin-utils", + "ryu", + "serde", + "time", +] + +[[package]] +name = "aws-smithy-types" +version = "1.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6dc683efb34b9e755675b37fedbe0103141e5b6df7bdc9eb6967756a8c167d8" +dependencies = [ + "base64-simd", + "bytes", + "bytes-utils", + "futures-core", + "http 0.2.12", + "http 1.5.0", + "http-body 0.4.6", + "http-body 1.1.0", + "http-body-util", + "itoa", + "num-integer", + "pin-project-lite", + "pin-utils", + "ryu", + "serde", + "time", + "tokio", + "tokio-util", +] + +[[package]] +name = "aws-smithy-xml" +version = "0.60.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ce02add1aa3677d022f8adf81dcbe3046a95f17a1b1e8979c145cd21d3d22b3" +dependencies = [ + "xmlparser", +] + +[[package]] +name = "aws-types" +version = "1.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "afb278e322f16f59630a83b6b2dc992a0b48aa74ed47b4130f193fae0053d713" +dependencies = [ + "aws-credential-types", + "aws-smithy-async 1.3.0", + "aws-smithy-runtime-api 1.14.0", + "aws-smithy-types 1.6.1", + "http 0.2.12", + "rustc_version", + "tracing", +] + +[[package]] +name = "base64" +version = "0.22.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" + +[[package]] +name = "base64-simd" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "339abbe78e73178762e23bea9dfd08e697eb3f3301cd4be981c0f78ba5859195" +dependencies = [ + "outref", + "vsimd", +] + +[[package]] +name = "bitflags" +version = "1.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" + +[[package]] +name = "bitflags" +version = "2.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" + +[[package]] +name = "block-buffer" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4152116fd6e9dadb291ae18fc1ec3575ed6d84c29642d97890f4b4a3417297e4" +dependencies = [ + "generic-array", +] + +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + +[[package]] +name = "block-buffer" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d2f6c7dbe95a6ed67ad9f18e57daf93a2f034c524b99fd2b76d18fdfeb6660aa" +dependencies = [ + "hybrid-array", +] + +[[package]] +name = "bs58" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bf88ba1141d185c399bee5288d850d63b8369520c1eafc32a0430b5b6c287bf4" +dependencies = [ + "tinyvec", +] + +[[package]] +name = "bumpalo" +version = "3.20.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" + +[[package]] +name = "byteorder" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" + +[[package]] +name = "bytes" +version = "1.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" + +[[package]] +name = "bytes-utils" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7dafe3a8757b027e2be6e4e5601ed563c55989fcf1546e933c66c8eb3a058d35" +dependencies = [ + "bytes", + "either", +] + +[[package]] +name = "cc" +version = "1.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5d262e149917187838d5b42777c8253bcb64500067342904e7d429499a6f277e" +dependencies = [ + "find-msvc-tools", + "shlex", +] + +[[package]] +name = "cfg-if" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" + +[[package]] +name = "chrono" +version = "0.4.45" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" +dependencies = [ + "iana-time-zone", + "num-traits", + "serde", + "windows-link", +] + +[[package]] +name = "clap" +version = "3.2.25" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4ea181bf566f71cb9a5d17a59e1871af638180a18fb0035c92ae62b705207123" +dependencies = [ + "atty", + "bitflags 1.3.2", + "clap_lex 0.2.4", + "indexmap 1.9.3", + "strsim 0.10.0", + "termcolor", + "textwrap", +] + +[[package]] +name = "clap" +version = "4.4.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e578d6ec4194633722ccf9544794b71b1385c3c027efe0c55db226fc880865c" +dependencies = [ + "clap_builder", + "clap_derive", +] + +[[package]] +name = "clap_builder" +version = "4.4.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4df4df40ec50c46000231c914968278b1eb05098cf8f1b3a518a95030e71d1c7" +dependencies = [ + "anstream", + "anstyle", + "clap_lex 0.6.0", + "strsim 0.10.0", +] + +[[package]] +name = "clap_derive" +version = "4.4.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf9804afaaf59a91e75b022a30fb7229a7901f60c755489cc61c9b423b836442" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "clap_lex" +version = "0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2850f2f5a82cbf437dd5af4d49848fbdfc27c157c3d010345776f952765261c5" +dependencies = [ + "os_str_bytes", +] + +[[package]] +name = "clap_lex" +version = "0.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "702fc72eb24e5a1e48ce58027a675bc24edd52096d5397d4aea7c6dd9eca0bd1" + +[[package]] +name = "cmov" +version = "0.5.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c9ea0ac24bc397ab3c98583a3c9ba74fa56b09a4449bbe172b9b1ddb016027a" + +[[package]] +name = "colorchoice" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" + +[[package]] +name = "const-oid" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a6ef517f0926dd24a1582492c791b6a4818a4d94e789a334894aa15b0d12f55c" + +[[package]] +name = "core-foundation" +version = "0.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b2a6cd9ae233e7f62ba4e9353e81a88df7fc8a5987b8d445b4d90c879bd156f6" +dependencies = [ + "core-foundation-sys", + "libc", +] + +[[package]] +name = "core-foundation-sys" +version = "0.8.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" + +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + +[[package]] +name = "cpufeatures" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b2a41393f66f16b0823bb79094d54ac5fbd34ab292ddafb9a0456ac9f87d201" +dependencies = [ + "libc", +] + +[[package]] +name = "crc" +version = "3.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5eb8a2a1cd12ab0d987a5d5e825195d372001a4094a0376319d5a0ad71c1ba0d" +dependencies = [ + "crc-catalog", +] + +[[package]] +name = "crc-catalog" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "217698eaf96b4a3f0bc4f3662aaa55bdf913cd54d7204591faa790070c6d0853" + +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "typenum", +] + +[[package]] +name = "crypto-common" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ce6e4c961d6cd6c9a86db418387425e8bdeaf05b3c8bc1411e6dca4c252f1453" +dependencies = [ + "hybrid-array", +] + +[[package]] +name = "ctutils" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7d5515a3834141de9eafb9717ad39eea8247b5674e6066c404e8c4b365d2a29e" +dependencies = [ + "cmov", +] + +[[package]] +name = "darling" +version = "0.23.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "25ae13da2f202d56bd7f91c25fba009e7717a1e4a1cc98a76d844b65ae912e9d" +dependencies = [ + "darling_core", + "darling_macro", +] + +[[package]] +name = "darling_core" +version = "0.23.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9865a50f7c335f53564bb694ef660825eb8610e0a53d3e11bf1b0d3df31e03b0" +dependencies = [ + "ident_case", + "proc-macro2", + "quote", + "strsim 0.11.1", + "syn 2.0.119", +] + +[[package]] +name = "darling_macro" +version = "0.23.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ac3984ec7bd6cfa798e62b4a642426a5be0e68f9401cfc2a01e3fa9ea2fcdb8d" +dependencies = [ + "darling_core", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "defmt" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e2953bfe4f93bbd20cc71198842756f77d161884c99ebbabc41d80231ded88d1" +dependencies = [ + "bitflags 1.3.2", + "defmt-macros", +] + +[[package]] +name = "defmt-macros" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bad9c72e7ca2137e0dc3813245a0d282fd6daad32fd800af018306a9169b5fe8" +dependencies = [ + "defmt-parser", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "defmt-parser" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" +dependencies = [ + "thiserror", +] + +[[package]] +name = "deranged" +version = "0.5.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" +dependencies = [ + "serde_core", +] + +[[package]] +name = "digest" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3dd60d1080a57a05ab032377049e0591415d2b31afd7028356dbf3cc6dcb066" +dependencies = [ + "generic-array", +] + +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer 0.10.4", + "crypto-common 0.1.7", +] + +[[package]] +name = "digest" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f1dd6dbb5841937940781866fa1281a1ff7bd3bf827091440879f9994983d5c2" +dependencies = [ + "block-buffer 0.12.1", + "const-oid", + "crypto-common 0.2.2", + "ctutils", +] + +[[package]] +name = "displaydoc" +version = "0.2.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "dyn-clone" +version = "1.0.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555" + +[[package]] +name = "eif_build" +version = "0.3.1" +dependencies = [ + "aws-nitro-enclaves-image-format 0.5.0", + "chrono", + "clap 4.4.18", + "serde", + "serde_json", + "sha2 0.9.9", +] + +[[package]] +name = "either" +version = "1.17.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9e5e8f6c15a24b9a3ee5efec809ccd006d3b30e8b3bb63c39af737c7f87daa1d" + +[[package]] +name = "equivalent" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" + +[[package]] +name = "errno" +version = "0.3.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" +dependencies = [ + "libc", + "windows-sys 0.61.2", +] + +[[package]] +name = "fastrand" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" + +[[package]] +name = "find-msvc-tools" +version = "0.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "26b73573e6edcd2af0cdf47bd6cb58f0b3839491263c314eaad1ccf24430e1de" + +[[package]] +name = "fnv" +version = "1.0.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" + +[[package]] +name = "foreign-types" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f6f339eb8adc052cd2ca78910fda869aefa38d22d5cb648e6485e4d3fc06f3b1" +dependencies = [ + "foreign-types-shared", +] + +[[package]] +name = "foreign-types-shared" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "00b0228411908ca8685dba7fc2cdd70ec9990a6e753e89b6ac91a84c40fbaf4b" + +[[package]] +name = "form_urlencoded" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" +dependencies = [ + "percent-encoding", +] + +[[package]] +name = "futures-channel" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "262590f4fe6afeb0bc83be1daa64e52657fe185690a958af7f3ad0e92085c5ae" +dependencies = [ + "futures-core", +] + +[[package]] +name = "futures-core" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" + +[[package]] +name = "futures-sink" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e34418ac499d6305c2fb5ad0ed2f6ac998c5f8ca209b4510f7f94242c647e307" + +[[package]] +name = "futures-task" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" + +[[package]] +name = "futures-util" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" +dependencies = [ + "futures-core", + "futures-task", + "pin-project-lite", + "slab", +] + +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + +[[package]] +name = "getrandom" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" +dependencies = [ + "cfg-if", + "libc", + "wasi", +] + +[[package]] +name = "getrandom" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" +dependencies = [ + "cfg-if", + "libc", + "r-efi", +] + +[[package]] +name = "h2" +version = "0.3.27" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0beca50380b1fc32983fc1cb4587bfa4bb9e78fc259aad4a0032d2080309222d" +dependencies = [ + "bytes", + "fnv", + "futures-core", + "futures-sink", + "futures-util", + "http 0.2.12", + "indexmap 2.14.0", + "slab", + "tokio", + "tokio-util", + "tracing", +] + +[[package]] +name = "h2" +version = "0.4.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6cb093c84e8bd9b188d4c4a8cb6579fc016968d14c99882163cd3ff402a4f155" +dependencies = [ + "atomic-waker", + "bytes", + "fnv", + "futures-core", + "futures-sink", + "http 1.5.0", + "indexmap 2.14.0", + "slab", + "tokio", + "tokio-util", + "tracing", +] + +[[package]] +name = "half" +version = "1.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1b43ede17f21864e81be2fa654110bf1e793774238d86ef8555c37e6519c0403" + +[[package]] +name = "hashbrown" +version = "0.12.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8a9ee70c43aaf417c914396645a0fa852624801b24ebb7ae78fe8272889ac888" + +[[package]] +name = "hashbrown" +version = "0.17.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" + +[[package]] +name = "heck" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "95505c38b4572b2d910cecb0281560f54b440a19336cbbcb27bf6ce6adc6f5a8" + +[[package]] +name = "hermit-abi" +version = "0.1.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "62b467343b94ba476dcb2500d242dadbb39557df889310ac77c5d99100aaac33" +dependencies = [ + "libc", +] + +[[package]] +name = "hex" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70" + +[[package]] +name = "hmac" +version = "0.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6303bc9732ae41b04cb554b844a762b4115a61bfaa81e3e83050991eeb56863f" +dependencies = [ + "digest 0.11.3", +] + +[[package]] +name = "http" +version = "0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "601cbb57e577e2f5ef5be8e7b83f0f63994f25aa94d673e54a92d5c516d101f1" +dependencies = [ + "bytes", + "fnv", + "itoa", +] + +[[package]] +name = "http" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" +dependencies = [ + "bytes", + "itoa", +] + +[[package]] +name = "http-body" +version = "0.4.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7ceab25649e9960c0311ea418d17bee82c0dcec1bd053b5f9a66e265a693bed2" +dependencies = [ + "bytes", + "http 0.2.12", + "pin-project-lite", +] + +[[package]] +name = "http-body" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" +dependencies = [ + "bytes", + "http 1.5.0", +] + +[[package]] +name = "http-body-util" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2" +dependencies = [ + "bytes", + "futures-core", + "http 1.5.0", + "http-body 1.1.0", + "pin-project-lite", +] + +[[package]] +name = "httparse" +version = "1.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" + +[[package]] +name = "httpdate" +version = "1.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9" + +[[package]] +name = "hybrid-array" +version = "0.4.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "707114b52a152fa7bdb290cd7cd5912d9467273b6d74e21b8d81aca1f8533f6b" +dependencies = [ + "typenum", +] + +[[package]] +name = "hyper" +version = "0.14.32" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "41dfc780fdec9373c01bae43289ea34c972e40ee3c9f6b3c8801a35f35586ce7" +dependencies = [ + "bytes", + "futures-channel", + "futures-core", + "futures-util", + "h2 0.3.27", + "http 0.2.12", + "http-body 0.4.6", + "httparse", + "httpdate", + "itoa", + "pin-project-lite", + "socket2 0.5.10", + "tokio", + "tower-service", + "tracing", + "want", +] + +[[package]] +name = "hyper-rustls" +version = "0.24.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ec3efd23720e2049821a693cbc7e65ea87c72f1c58ff2f9522ff332b1491e590" +dependencies = [ + "futures-util", + "http 0.2.12", + "hyper", + "log", + "rustls", + "tokio", + "tokio-rustls", +] + +[[package]] +name = "iana-time-zone" +version = "0.1.65" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e31bc9ad994ba00e440a8aa5c9ef0ec67d5cb5e5cb0cc7f8b744a35b389cc470" +dependencies = [ + "android_system_properties", + "core-foundation-sys", + "iana-time-zone-haiku", + "js-sys", + "log", + "wasm-bindgen", + "windows-core", +] + +[[package]] +name = "iana-time-zone-haiku" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" +dependencies = [ + "cc", +] + +[[package]] +name = "icu_collections" +version = "2.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4c6b649701667bbe825c3b7e6388cb521c23d88644678e83c0c4d0a621a34b43" +dependencies = [ + "displaydoc", + "potential_utf", + "yoke", + "zerofrom", + "zerovec", +] + +[[package]] +name = "icu_locale_core" +version = "2.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "edba7861004dd3714265b4db54a3c390e880ab658fec5f7db895fae2046b5bb6" +dependencies = [ + "displaydoc", + "litemap 0.8.2", + "tinystr", + "writeable", + "zerovec", +] + +[[package]] +name = "icu_normalizer" +version = "2.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5f6c8828b67bf8908d82127b2054ea1b4427ff0230ee9141c54251934ab1b599" +dependencies = [ + "icu_collections", + "icu_normalizer_data", + "icu_properties", + "icu_provider", + "smallvec", + "zerovec", +] + +[[package]] +name = "icu_normalizer_data" +version = "2.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7aedcccd01fc5fe81e6b489c15b247b8b0690feb23304303a9e560f37efc560a" + +[[package]] +name = "icu_properties" +version = "2.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "020bfc02fe870ec3a66d93e677ccca0562506e5872c650f893269e08615d74ec" +dependencies = [ + "icu_collections", + "icu_locale_core", + "icu_properties_data", + "icu_provider", + "zerotrie", + "zerovec", +] + +[[package]] +name = "icu_properties_data" +version = "2.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "616c294cf8d725c6afcd8f55abc17c56464ef6211f9ed59cccffe534129c77af" + +[[package]] +name = "icu_provider" +version = "2.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85962cf0ce02e1e0a629cc34e7ca3e373ce20dda4c4d7294bbd0bf1fdb59e614" +dependencies = [ + "displaydoc", + "icu_locale_core", + "writeable", + "yoke", + "zerofrom", + "zerotrie", + "zerovec", +] + +[[package]] +name = "ident_case" +version = "1.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9e0384b61958566e926dc50660321d12159025e767c18e043daf26b70104c39" + +[[package]] +name = "idna" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3b0875f23caa03898994f6ddc501886a45c7d3d62d04d2d90788d47be1b1e4de" +dependencies = [ + "idna_adapter", + "smallvec", + "utf8_iter", +] + +[[package]] +name = "idna_adapter" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3acae9609540aa318d1bc588455225fb2085b9ed0c4f6bd0d9d5bcd86f1a0344" +dependencies = [ + "icu_normalizer", + "icu_properties", +] + +[[package]] +name = "indexmap" +version = "1.9.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bd070e393353796e801d209ad339e89596eb4c8d430d18ede6a1cced8fafbd99" +dependencies = [ + "autocfg", + "hashbrown 0.12.3", + "serde", +] + +[[package]] +name = "indexmap" +version = "2.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d466e9454f08e4a911e14806c24e16fba1b4c121d1ea474396f396069cf949d9" +dependencies = [ + "equivalent", + "hashbrown 0.17.1", + "serde", + "serde_core", +] + +[[package]] +name = "is_terminal_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695" + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "jiff" +version = "0.2.35" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "668b7183bd07af9a4885f5c35b0cc5c83c4607a913c16b7e17291832910d2dcc" +dependencies = [ + "defmt", + "jiff-core", + "jiff-static", + "jiff-tzdb-platform", + "log", + "portable-atomic", + "portable-atomic-util", + "serde_core", + "windows-link", +] + +[[package]] +name = "jiff-core" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7feca88439efe53da3754500c1851dedf3cb36c524dd5cf8225cc0794de95d09" +dependencies = [ + "defmt", +] + +[[package]] +name = "jiff-static" +version = "0.2.35" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a69dcb3a21cfb32ce1cd056169337ca284af0766dd766e7878819b251a49204" +dependencies = [ + "jiff-core", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "jiff-tzdb" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "142bd39932ad231f10513df9ab62661fead8719872150b7ad02a2df79f4e141e" + +[[package]] +name = "jiff-tzdb-platform" +version = "0.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "875a5a69ac2bab1a891711cf5eccbec1ce0341ea805560dcd90b7a2e925132e8" +dependencies = [ + "jiff-tzdb", +] + +[[package]] +name = "js-sys" +version = "0.3.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a" +dependencies = [ + "cfg-if", + "futures-util", + "wasm-bindgen", +] + +[[package]] +name = "libc" +version = "0.2.189" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" + +[[package]] +name = "linux-raw-sys" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" + +[[package]] +name = "litemap" +version = "0.7.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4ee93343901ab17bd981295f2cf0026d4ad018c7c31ba84549a4ddbb47a45104" + +[[package]] +name = "litemap" +version = "0.8.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" + +[[package]] +name = "log" +version = "0.4.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "mio" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "30d65c71f1ce40ab09135ce117d742b9f8a19ff91a41a8b57ed50bc2de59c427" +dependencies = [ + "libc", + "wasi", + "windows-sys 0.61.2", +] + +[[package]] +name = "num-conv" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "521739c6d2bac4aa25192232afe6841231376b2b26d4d9fae5ecf8ca5772e441" + +[[package]] +name = "num-derive" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed3955f1a9c7c0c15e092f9c887db08b1fc683305fdf6eb6684f22555355e202" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "num-integer" +version = "0.1.46" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7969661fd2958a5cb096e56c8e1ad0444ac2bbcd0061bd28660485a44879858f" +dependencies = [ + "num-traits", +] + +[[package]] +name = "num-traits" +version = "0.2.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" +dependencies = [ + "autocfg", +] + +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + +[[package]] +name = "once_cell_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" + +[[package]] +name = "opaque-debug" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c08d65885ee38876c4f86fa503fb49d7b507c2b62552df7c70b2fce627e06381" + +[[package]] +name = "openssl" +version = "0.10.81" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "77823a27f0babb03091cb9ed9ef80af3b39dbc82f97e8fa530374b7dafd87a45" +dependencies = [ + "bitflags 2.13.1", + "cfg-if", + "foreign-types", + "libc", + "openssl-macros", + "openssl-sys", +] + +[[package]] +name = "openssl-macros" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a948666b637a0f465e8564c73e89d4dde00d72d4d473cc972f390fc3dcee7d9c" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "openssl-probe" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7c87def4c32ab89d880effc9e097653c8da5d6ef28e6b539d313baaacfbafcbe" + +[[package]] +name = "openssl-sys" +version = "0.9.117" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b47e7e6bb2c38cd930d25a23b40fa52e068c10e85f3e03a7f5ba5aaca5713695" +dependencies = [ + "cc", + "libc", + "pkg-config", + "vcpkg", +] + +[[package]] +name = "os_str_bytes" +version = "6.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e2355d85b9a3786f481747ced0e0ff2ba35213a1f9bd406ed906554d7af805a1" + +[[package]] +name = "outref" +version = "0.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1a80800c0488c3a21695ea981a54918fbb37abf04f4d0720c453632255e2ff0e" + +[[package]] +name = "percent-encoding" +version = "2.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" + +[[package]] +name = "pin-project-lite" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" + +[[package]] +name = "pin-utils" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b870d8c151b6f2fb93e84a13146138f05d02ed11c7e7c54f8826aaaf7c9f184" + +[[package]] +name = "pkg-config" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" + +[[package]] +name = "portable-atomic" +version = "1.15.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" + +[[package]] +name = "portable-atomic-util" +version = "0.2.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c2a106d1259c23fac8e543272398ae0e3c0b8d33c88ed73d0cc71b0f1d902618" +dependencies = [ + "portable-atomic", +] + +[[package]] +name = "potential_utf" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b73949432f5e2a09657003c25bca5e19a0e9c84f8058ca374f49e0ebe605af77" +dependencies = [ + "zerovec", +] + +[[package]] +name = "powerfmt" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "439ee305def115ba05938db6eb1644ff94165c5ab5e9420d1c1bcedbba909391" + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "r-efi" +version = "6.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" + +[[package]] +name = "ref-cast" +version = "1.0.26" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "216e8f773d7923bcba9ceb86a86c93cabb3903a11872fc3f138c49630e50b96d" +dependencies = [ + "ref-cast-impl", +] + +[[package]] +name = "ref-cast-impl" +version = "1.0.26" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2c9283685feec7d69af75fb0e858d5e7378f33fe4fc699383b2916ab9273e03c" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "regex" +version = "1.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d" +dependencies = [ + "aho-corasick", + "memchr", + "regex-automata", + "regex-syntax", +] + +[[package]] +name = "regex-automata" +version = "0.4.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" +dependencies = [ + "aho-corasick", + "memchr", + "regex-syntax", +] + +[[package]] +name = "regex-lite" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cab834c73d247e67f4fae452806d17d3c7501756d98c8808d7c9c7aa7d18f973" + +[[package]] +name = "regex-syntax" +version = "0.8.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" + +[[package]] +name = "ring" +version = "0.17.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7" +dependencies = [ + "cc", + "cfg-if", + "getrandom 0.2.17", + "libc", + "untrusted", + "windows-sys 0.52.0", +] + +[[package]] +name = "rustc_version" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cfcb3a22ef46e85b45de6ee7e79d063319ebb6594faafcf1c225ea92ab6e9b92" +dependencies = [ + "semver", +] + +[[package]] +name = "rustix" +version = "1.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190" +dependencies = [ + "bitflags 2.13.1", + "errno", + "libc", + "linux-raw-sys", + "windows-sys 0.61.2", +] + +[[package]] +name = "rustls" +version = "0.21.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f56a14d1f48b391359b22f731fd4bd7e43c97f3c50eee276f3aa09c94784d3e" +dependencies = [ + "log", + "ring", + "rustls-webpki", + "sct", +] + +[[package]] +name = "rustls-native-certs" +version = "0.8.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dab5152771c58876a2146916e53e35057e1a4dfa2b9df0f0305b07f611fdea4d" +dependencies = [ + "openssl-probe", + "rustls-pki-types", + "schannel", + "security-framework", +] + +[[package]] +name = "rustls-pki-types" +version = "1.15.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" +dependencies = [ + "zeroize", +] + +[[package]] +name = "rustls-webpki" +version = "0.101.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b6275d1ee7a1cd780b64aca7726599a1dbc893b1e64144529e55c3c2f745765" +dependencies = [ + "ring", + "untrusted", +] + +[[package]] +name = "rustversion" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" + +[[package]] +name = "ryu" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" + +[[package]] +name = "schannel" +version = "0.1.29" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "91c1b7e4904c873ef0710c1f407dde2e6287de2bebc1bbbf7d430bb7cbffd939" +dependencies = [ + "windows-sys 0.61.2", +] + +[[package]] +name = "schemars" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4cd191f9397d57d581cddd31014772520aa448f65ef991055d7f61582c65165f" +dependencies = [ + "dyn-clone", + "ref-cast", + "serde", + "serde_json", +] + +[[package]] +name = "schemars" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "687274d293b6cdc6e73e0fee520bf2049650090d7164f87672d212a3c530cf4a" +dependencies = [ + "dyn-clone", + "ref-cast", + "serde", + "serde_json", +] + +[[package]] +name = "sct" +version = "0.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da046153aa2352493d6cb7da4b6e5c0c057d8a1d0a9aa8560baffdd945acd414" +dependencies = [ + "ring", + "untrusted", +] + +[[package]] +name = "security-framework" +version = "3.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" +dependencies = [ + "bitflags 2.13.1", + "core-foundation", + "core-foundation-sys", + "libc", + "security-framework-sys", +] + +[[package]] +name = "security-framework-sys" +version = "2.17.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ce2691df843ecc5d231c0b14ece2acc3efb62c0a398c7e1d875f3983ce020e3" +dependencies = [ + "core-foundation-sys", + "libc", +] + +[[package]] +name = "semver" +version = "1.0.28" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", + "serde_derive", +] + +[[package]] +name = "serde_bytes" +version = "0.11.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a5d440709e79d88e51ac01c4b72fc6cb7314017bb7da9eeff678aa94c10e3ea8" +dependencies = [ + "serde", + "serde_core", +] + +[[package]] +name = "serde_cbor" +version = "0.11.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2bef2ebfde456fb76bbcf9f59315333decc4fda0b2b44b420243c11e0f5ec1f5" +dependencies = [ + "half", + "serde", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "serde_repr" +version = "0.1.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8d3b1629de253c70a0508c3899572da79ca359fdab27c7920ff00406df418906" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "serde_with" +version = "3.22.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ee78f1fbe43ac4a0e47aadb3dbd357b69eb0d3793e948624cd03dd2750ab1c0a" +dependencies = [ + "base64", + "bs58", + "chrono", + "hex", + "indexmap 1.9.3", + "indexmap 2.14.0", + "jiff", + "schemars 0.9.0", + "schemars 1.2.2", + "serde_core", + "serde_json", + "serde_with_macros", + "time", +] + +[[package]] +name = "serde_with_macros" +version = "3.22.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8705578779c2b6bd90d84d66eb2e206b708b1a4d7b9f17641b293545bf1c7e46" +dependencies = [ + "darling", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "sha2" +version = "0.9.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4d58a1e1bf39749807d89cf2d98ac2dfa0ff1cb3faa38fbb64dd88ac8013d800" +dependencies = [ + "block-buffer 0.9.0", + "cfg-if", + "cpufeatures 0.2.17", + "digest 0.9.0", + "opaque-debug", +] + +[[package]] +name = "sha2" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +dependencies = [ + "cfg-if", + "cpufeatures 0.2.17", + "digest 0.10.7", +] + +[[package]] +name = "sha2" +version = "0.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "446ba717509524cb3f22f17ecc096f10f4822d76ab5c0b9822c5f9c284e825f4" +dependencies = [ + "cfg-if", + "cpufeatures 0.3.0", + "digest 0.11.3", +] + +[[package]] +name = "shlex" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" + +[[package]] +name = "signal-hook-registry" +version = "1.4.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c4db69cba1110affc0e9f7bcd48bbf87b3f4fc7c61fc9155afd4c469eb3d6c1b" +dependencies = [ + "errno", + "libc", +] + +[[package]] +name = "slab" +version = "0.4.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" + +[[package]] +name = "smallvec" +version = "1.15.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" + +[[package]] +name = "socket2" +version = "0.5.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e22376abed350d73dd1cd119b57ffccad95b4e585a7cda43e286245ce23c0678" +dependencies = [ + "libc", + "windows-sys 0.52.0", +] + +[[package]] +name = "socket2" +version = "0.6.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" +dependencies = [ + "libc", + "windows-sys 0.61.2", +] + +[[package]] +name = "stable_deref_trait" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" + +[[package]] +name = "strsim" +version = "0.10.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "73473c0e59e6d5812c5dfe2a064a6444949f089e20eec9a2e5506596494e4623" + +[[package]] +name = "strsim" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" + +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "synstructure" +version = "0.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "tempfile" +version = "3.27.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" +dependencies = [ + "fastrand", + "getrandom 0.4.3", + "once_cell", + "rustix", + "windows-sys 0.61.2", +] + +[[package]] +name = "termcolor" +version = "1.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "06794f8f6c5c898b3275aebefa6b8a1cb24cd2c6c79397ab15774837a0bc5755" +dependencies = [ + "winapi-util", +] + +[[package]] +name = "textwrap" +version = "0.16.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c13547615a44dc9c452a8a534638acdf07120d4b6847c8178705da06306a3057" + +[[package]] +name = "thiserror" +version = "2.0.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" +dependencies = [ + "thiserror-impl", +] + +[[package]] +name = "thiserror-impl" +version = "2.0.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "time" +version = "0.3.55" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cdb87b95ec50ddfa440816d227a17b2ccbdda963a316a727fda0fc4334f7d134" +dependencies = [ + "deranged", + "num-conv", + "powerfmt", + "serde_core", + "time-core", + "time-macros", +] + +[[package]] +name = "time-core" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9e1c906769ad99c88eaa54e728060edef082f8e358ff32030cb7c7d315e81109" + +[[package]] +name = "time-macros" +version = "0.2.32" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7e689342a48d2ea927c87ea50cabf8594854bf940e9310208848d680d668ed85" +dependencies = [ + "num-conv", + "time-core", +] + +[[package]] +name = "tinystr" +version = "0.8.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "42d3e9c45c09de15d06dd8acf5f4e0e399e85927b7f00711024eb7ae10fa4869" +dependencies = [ + "displaydoc", + "zerovec", +] + +[[package]] +name = "tinyvec" +version = "1.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb4ebadaa0af04fab11ae01eb5f9fdb5f9c5b875506e210e71c07873528baa7f" +dependencies = [ + "tinyvec_macros", +] + +[[package]] +name = "tinyvec_macros" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" + +[[package]] +name = "tokio" +version = "1.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" +dependencies = [ + "bytes", + "libc", + "mio", + "pin-project-lite", + "signal-hook-registry", + "socket2 0.6.5", + "tokio-macros", + "windows-sys 0.61.2", +] + +[[package]] +name = "tokio-macros" +version = "2.7.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "tokio-rustls" +version = "0.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c28327cf380ac148141087fbfb9de9d7bd4e84ab5d2c28fbc911d753de8a7081" +dependencies = [ + "rustls", + "tokio", +] + +[[package]] +name = "tokio-util" +version = "0.7.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" +dependencies = [ + "bytes", + "futures-core", + "futures-sink", + "libc", + "pin-project-lite", + "tokio", +] + +[[package]] +name = "tower-service" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3" + +[[package]] +name = "tracing" +version = "0.1.44" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "63e71662fa4b2a2c3a26f570f037eb95bb1f85397f3cd8076caed2f026a6d100" +dependencies = [ + "pin-project-lite", + "tracing-attributes", + "tracing-core", +] + +[[package]] +name = "tracing-attributes" +version = "0.1.31" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "tracing-core" +version = "0.1.36" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" +dependencies = [ + "once_cell", +] + +[[package]] +name = "try-lock" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" + +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + +[[package]] +name = "unicode-ident" +version = "1.0.24" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" + +[[package]] +name = "untrusted" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" + +[[package]] +name = "url" +version = "2.5.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff67a8a4397373c3ef660812acab3268222035010ab8680ec4215f38ba3d0eed" +dependencies = [ + "form_urlencoded", + "idna", + "percent-encoding", + "serde", +] + +[[package]] +name = "urlencoding" +version = "2.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "daf8dba3b7eb870caf1ddeed7bc9d2a049f3cfdfae7cb521b087cc33ae4c49da" + +[[package]] +name = "utf8_iter" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" + +[[package]] +name = "utf8parse" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" + +[[package]] +name = "uuid" +version = "1.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bf3923a6f5c4c6382e0b653c4117f48d631ea17f38ed86e2a828e6f7412f5239" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "vcpkg" +version = "0.2.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + +[[package]] +name = "vsimd" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5c3082ca00d5a5ef149bb8b555a72ae84c9c59f7250f013ac822ac2e49b19c64" + +[[package]] +name = "want" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bfa7760aed19e106de2c7c0b581b509f2f25d3dacaf737cb82ac61bc6d760b0e" +dependencies = [ + "try-lock", +] + +[[package]] +name = "wasi" +version = "0.11.1+wasi-snapshot-preview1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" + +[[package]] +name = "wasm-bindgen" +version = "0.2.127" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70" +dependencies = [ + "cfg-if", + "once_cell", + "rustversion", + "wasm-bindgen-macro", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-macro" +version = "0.2.127" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1" +dependencies = [ + "quote", + "wasm-bindgen-macro-support", +] + +[[package]] +name = "wasm-bindgen-macro-support" +version = "0.2.127" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284" +dependencies = [ + "bumpalo", + "proc-macro2", + "quote", + "syn 2.0.119", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-shared" +version = "0.2.127" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "winapi" +version = "0.3.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5c839a674fcd7a98952e593242ea400abe93992746761e38641405d28b00f419" +dependencies = [ + "winapi-i686-pc-windows-gnu", + "winapi-x86_64-pc-windows-gnu", +] + +[[package]] +name = "winapi-i686-pc-windows-gnu" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ac3b87c63620426dd9b991e5ce0329eff545bccbbb34f3be09ff6fb6ab51b7b6" + +[[package]] +name = "winapi-util" +version = "0.1.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22" +dependencies = [ + "windows-sys 0.61.2", +] + +[[package]] +name = "winapi-x86_64-pc-windows-gnu" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f" + +[[package]] +name = "windows-core" +version = "0.62.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" +dependencies = [ + "windows-implement", + "windows-interface", + "windows-link", + "windows-result", + "windows-strings", +] + +[[package]] +name = "windows-implement" +version = "0.60.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "windows-interface" +version = "0.59.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-result" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-strings" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-sys" +version = "0.52.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" +dependencies = [ + "windows-targets", +] + +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-targets" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" +dependencies = [ + "windows_aarch64_gnullvm", + "windows_aarch64_msvc", + "windows_i686_gnu", + "windows_i686_gnullvm", + "windows_i686_msvc", + "windows_x86_64_gnu", + "windows_x86_64_gnullvm", + "windows_x86_64_msvc", +] + +[[package]] +name = "windows_aarch64_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" + +[[package]] +name = "windows_aarch64_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" + +[[package]] +name = "windows_i686_gnu" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" + +[[package]] +name = "windows_i686_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" + +[[package]] +name = "windows_i686_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" + +[[package]] +name = "windows_x86_64_gnu" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" + +[[package]] +name = "windows_x86_64_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" + +[[package]] +name = "windows_x86_64_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" + +[[package]] +name = "writeable" +version = "0.6.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4" + +[[package]] +name = "xmlparser" +version = "0.13.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "66fee0b777b0f5ac1c69bb06d361268faafa61cd4682ae064a171c16c433e9e4" + +[[package]] +name = "yoke" +version = "0.8.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72d6e5c6afb84d73944e5cedb052c4680d5657337201555f9f2a16b7406d4954" +dependencies = [ + "stable_deref_trait", + "yoke-derive", + "zerofrom", +] + +[[package]] +name = "yoke-derive" +version = "0.8.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", + "synstructure", +] + +[[package]] +name = "zerofrom" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cff3ee08c995dee1859d998dea82f7374f2826091dd9cd47def953cae446cd2e" +dependencies = [ + "zerofrom-derive", +] + +[[package]] +name = "zerofrom-derive" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", + "synstructure", +] + +[[package]] +name = "zeroize" +version = "1.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" + +[[package]] +name = "zerotrie" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2a59c17a5562d507e4b54960e8569ebee33bee890c70aa3fe7b97e85a9fd7851" +dependencies = [ + "displaydoc", + "yoke", + "zerofrom", +] + +[[package]] +name = "zerovec" +version = "0.11.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6c28719294829477f525be0186d13efa9a3c602f7ec202ca9e353d310fb9a002" +dependencies = [ + "yoke", + "zerofrom", + "zerovec-derive", +] + +[[package]] +name = "zerovec-derive" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/deploy/parent/run-parent.sh b/deploy/parent/run-parent.sh new file mode 100755 index 0000000..482dc5a --- /dev/null +++ b/deploy/parent/run-parent.sh @@ -0,0 +1,87 @@ +#!/usr/bin/env bash +# What the EC2 parent instance runs so an enclave can reach, and be reached by, +# the network. +# +# An enclave has no NIC. Everything it sends leaves over vsock and arrives +# here, so nothing in the enclave works — not S3, not ACME, not DNS, not an +# inbound request — until these are running. An enclave that boots and then +# answers nothing is almost always this. +# +# internet :443 +# │ +# ▼ +# ┌─────────────────────────────────────────────┐ parent (untrusted) +# │ gvproxy --listen vsock://:1024 │ +# │ expose :443 → 192.168.127.2:443 │ +# └─────────────────────────────────────────────┘ +# │ AF_VSOCK, ethernet frames +# ▼ +# ┌─────────────────────────────────────────────┐ enclave (attested) +# │ gvforwarder → tap0 192.168.127.2 │ +# │ enclave-runtime, TLS terminated here │ +# └─────────────────────────────────────────────┘ +# +# The parent carries ciphertext it cannot read: TLS to S3, KMS and the ACME +# provider is established inside the enclave and terminated at the far end, and +# inbound HTTPS is terminated by the enclave itself. What the parent does see +# is metadata — who is talked to, when, how much — and it can of course refuse +# to carry anything at all. Neither is new; it already decides whether the +# enclave runs. +set -euo pipefail + +VSOCK_PORT="${VSOCK_PORT:-1024}" +ENCLAVE_IP="${ENCLAVE_IP:-192.168.127.2}" +API_SOCKET="${API_SOCKET:-/tmp/gvproxy-network.sock}" +# Ports to forward from this instance into the enclave. 443 is the HTTPS +# listener; ACME's TLS-ALPN-01 challenge arrives on the same one, which is why +# no second port is needed. +FORWARD_PORTS="${FORWARD_PORTS:-443}" + +command -v gvproxy >/dev/null || { + cat >&2 <<'EOF' +gvproxy not found. Install it from containers/gvisor-tap-vsock: + + go install github.com/containers/gvisor-tap-vsock/cmd/gvproxy@latest + +or take a release binary. The matching `gvforwarder` goes *inside* the enclave +image, not here — see deploy/Dockerfile. +EOF + exit 1 +} + +rm -f "$API_SOCKET" + +echo "== starting gvproxy on vsock port $VSOCK_PORT ==" +gvproxy \ + --listen "vsock://:${VSOCK_PORT}" \ + --listen "unix://${API_SOCKET}" \ + & +GVPROXY_PID=$! +trap 'kill "$GVPROXY_PID" 2>/dev/null || true' EXIT + +for _ in $(seq 1 50); do + [[ -S "$API_SOCKET" ]] && break + sleep 0.2 +done +[[ -S "$API_SOCKET" ]] || { echo "gvproxy never opened its API socket" >&2; exit 1; } + +# Inbound forwarding is a runtime call rather than a flag: gvproxy exposes it +# over the same API socket, and doing it after startup means the enclave can be +# restarted without restarting the proxy. +for port in $FORWARD_PORTS; do + echo "== forwarding :$port → ${ENCLAVE_IP}:$port ==" + curl -sf --unix-socket "$API_SOCKET" \ + http://localhost/services/forwarder/expose \ + -X POST \ + -H 'Content-Type: application/json' \ + -d "{\"local\":\":${port}\",\"remote\":\"${ENCLAVE_IP}:${port}\"}" \ + || { echo "failed to expose port $port" >&2; exit 1; } +done + +echo +echo "gvproxy is up. Start the enclave with:" +echo " nitro-cli run-enclave --eif-path s3fs.eif --cpu-count 2 --memory 2048" +echo +echo "Forwarded: $FORWARD_PORTS → $ENCLAVE_IP" +echo "API: $API_SOCKET" +wait "$GVPROXY_PID" diff --git a/deploy/ptp-check.sh b/deploy/ptp-check.sh new file mode 100755 index 0000000..69c9936 --- /dev/null +++ b/deploy/ptp-check.sh @@ -0,0 +1,110 @@ +#!/usr/bin/env bash +# Verify that enclave-runtime can read a PTP hardware clock. +# +# /dev/ptp0 is root-owned, so this needs a context where we are root. Two are +# offered, and they prove different things: +# +# container Passes the host's real PHC into a container via --device. This +# is the stronger test: the PHC is genuinely a different clock +# from CLOCK_REALTIME, so a non-zero skew proves we are reading +# the device rather than falling through to the system clock. +# Needs membership of the `docker` group. +# +# qemu Boots a VM and uses ptp_kvm to expose a /dev/ptp0 inside it. +# Exercises the same device interface, but the clock behind +# ptp_kvm *is* the host's clock, so the skew is ~0 and it cannot +# distinguish "read the PHC" from "read CLOCK_REALTIME". Useful +# where no PHC exists, or where docker is unavailable. +# +# Neither emulates Nitro. Full fidelity needs QEMU >= 9.1's `nitro-enclave` +# machine and an EIF, which belongs with NSM and KMS in M8. +# +# ./deploy/ptp-check.sh # container (default) +# ./deploy/ptp-check.sh --mode qemu +set -euo pipefail + +MODE=container +DEVICE=/dev/ptp0 +while [[ $# -gt 0 ]]; do + case "$1" in + --mode) MODE="$2"; shift 2 ;; + --device) DEVICE="$2"; shift 2 ;; + *) echo "unknown argument: $1" >&2; exit 2 ;; + esac +done + +REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +BIN="$REPO/target/release/enclave-runtime" +# clock-check never mounts anything, but the CLI requires these. +DUMMY_ARGS=(--clock-check --clock-source ptp --bucket unused + --master-key 00000000000000000000000000000000000000000000000000000000000000ab) + +[[ -x "$BIN" ]] || { echo "build it first: cargo build --release -p enclave-runtime" >&2; exit 1; } + +case "$MODE" in +container) + command -v docker >/dev/null || { echo "docker not found" >&2; exit 1; } + [[ -e "$DEVICE" ]] || { echo "$DEVICE does not exist on this host" >&2; exit 1; } + + echo "== PTP via container, device $DEVICE ==" + # The AWS SDK's TLS provider panics at construction when it finds no trust + # store, even for a plain-HTTP endpoint, so the host's certs come along. + docker run --rm --device "$DEVICE" \ + -v /etc/ssl/certs:/etc/ssl/certs:ro \ + -v "$BIN:/enclave-runtime:ro" \ + ubuntu:24.04 /enclave-runtime "${DUMMY_ARGS[@]}" --ptp-device "$DEVICE" + ;; + +qemu) + command -v qemu-system-x86_64 >/dev/null || { + echo "qemu-system-x86_64 not found: apt install qemu-system-x86" >&2; exit 1; } + command -v cloud-localds >/dev/null || { + echo "cloud-localds not found: apt install cloud-image-utils" >&2; exit 1; } + [[ -w /dev/kvm ]] || { + echo "/dev/kvm is not writable. ptp_kvm needs KVM; TCG emulation will" >&2 + echo "not produce the device, so this mode cannot work without it." >&2 + exit 1; } + + WORK="${TMPDIR:-/tmp}/s3fs-ptp-qemu" + mkdir -p "$WORK" + IMG="$WORK/noble.img" + SEED="$WORK/seed.iso" + + if [[ ! -f "$IMG" ]]; then + echo "== fetching Ubuntu cloud image ==" + curl -sSL -o "$WORK/noble.orig.img" \ + https://cloud-images.ubuntu.com/noble/current/noble-server-cloudimg-amd64.img + cp "$WORK/noble.orig.img" "$IMG" + qemu-img resize "$IMG" +2G + fi + + # cloud-init loads ptp_kvm, runs the check, prints a sentinel, powers off. + # Everything runs as root, which is the whole reason for the VM. + cat > "$WORK/user-data" < "$WORK/meta-data" + cloud-localds "$SEED" "$WORK/user-data" "$WORK/meta-data" + + echo "== booting VM; ptp_kvm exposes the host clock as /dev/ptp0 inside ==" + qemu-system-x86_64 \ + -enable-kvm -m 1024 -smp 2 -nographic \ + -drive "file=$IMG,format=qcow2,if=virtio" \ + -drive "file=$SEED,format=raw,if=virtio" \ + -virtfs "local,path=$(dirname "$BIN"),mount_tag=host,security_model=mapped-xattr" \ + | tee "$WORK/console.log" + + grep -q "PTP-CHECK-DONE" "$WORK/console.log" || { echo "VM did not finish" >&2; exit 1; } + ! grep -q "PTP-CHECK-FAIL" "$WORK/console.log" || { echo "check failed, see $WORK/console.log" >&2; exit 1; } + echo "== ok ==" + ;; + +*) + echo "unknown mode: $MODE (expected container or qemu)" >&2; exit 2 ;; +esac diff --git a/deploy/qemu-nitro/Dockerfile b/deploy/qemu-nitro/Dockerfile new file mode 100644 index 0000000..90da77a --- /dev/null +++ b/deploy/qemu-nitro/Dockerfile @@ -0,0 +1,46 @@ +# QEMU with the nitro-enclave machine, plus the tooling to build an EIF. +# +# Distro packages are not usable: Debian sid ships QEMU 11 and the machine type +# is present, but `-device help` has no virtio-nsm, because the NSM device is +# only compiled when libcbor and gnutls are found at configure time. Without it +# the machine boots but has no Nitro Security Module, which is the one thing +# this harness exists to exercise. So QEMU is built from source with both. +# +# Built in a container rather than on the host so it needs no root there, and +# so the toolchain is pinned rather than whatever the developer happens to have. +FROM debian:trixie + +RUN apt-get update && apt-get install -y --no-install-recommends \ + git ca-certificates curl xz-utils cpio \ + build-essential ninja-build meson pkg-config python3 python3-venv \ + libglib2.0-dev libpixman-1-dev libslirp-dev \ + libcbor-dev libgnutls28-dev \ + && rm -rf /var/lib/apt/lists/* + +ARG QEMU_TAG=v9.2.0 +RUN git clone --depth 1 --branch ${QEMU_TAG} https://gitlab.com/qemu-project/qemu.git /qemu + +WORKDIR /qemu/build +# Only x86_64 softmmu: a full target list triples the build for nothing. +RUN ../configure \ + --target-list=x86_64-softmmu \ + --enable-kvm \ + --enable-gnutls \ + --disable-docs --disable-guest-agent --disable-sdl --disable-gtk \ + --disable-vnc --disable-werror \ + && ninja -j"$(nproc)" \ + && ninja install + +# Fail the build rather than the run if the NSM device is missing: that is the +# single thing this image is for, and a silent absence would be discovered much +# later, inside a boot that hangs for unrelated-looking reasons. +RUN qemu-system-x86_64 -device help 2>&1 | grep -q virtio-nsm \ + || (echo "FATAL: QEMU built without virtio-nsm; libcbor/gnutls were not detected" >&2; exit 1) \ + && qemu-system-x86_64 -M help | grep -iq nitro \ + || (echo "FATAL: QEMU built without the nitro-enclave machine" >&2; exit 1) + +RUN qemu-system-x86_64 --version | head -1 > /qemu-version \ + && qemu-system-x86_64 -M help | grep -i nitro >> /qemu-version \ + && qemu-system-x86_64 -device help 2>&1 | grep -i nsm >> /qemu-version + +WORKDIR /work diff --git a/deploy/qemu-nitro/dev-enclave.sh b/deploy/qemu-nitro/dev-enclave.sh new file mode 100755 index 0000000..16155c8 --- /dev/null +++ b/deploy/qemu-nitro/dev-enclave.sh @@ -0,0 +1,376 @@ +#!/usr/bin/env bash +# A real enclave, on this machine, for developing a client against. +# +# Same stack as the e2e — the same emulated Nitro machine, the same ACME +# issuance from a real CA, the same block store, the same attested boot — but +# it stays up, and it will run *your* guest. The e2e brings this up to assert +# against it and then tears it down; this brings it up and gets out of the way. +# +# deploy/qemu-nitro/dev-enclave.sh --guest path/to/component.wasm +# +# It prints what a client needs to talk to it, and then waits. Ctrl-C stops +# everything it started. +# +# --------------------------------------------------------------------------- +# What is real here, and what is not +# +# Real: the enclave boots as an EIF under QEMU's nitro-enclave machine, takes +# its address by DHCP over emulated vsock, mounts the encrypted block store over +# that link, fetches your component from the store and measures it into PCR16 +# before it can obtain a key, orders a certificate from a CA over RFC 8555 +# TLS-ALPN-01, and gates every request on a WebAuthn assertion bound to that +# request. A client verifies the attestation document by signature, certificate +# chain, validity window, pinned root and both measurements — the same code path +# with the same flags it will use against hardware. +# +# Not real: the key that signs those documents. QEMU's NSM does not sign +# anything, so this image mints a chain at boot and re-signs what the device +# produced. The key is inside an image you control, so a verified document here +# means "this image said so" and not "a Nitro enclave said so". The root changes +# every boot for that reason — there is deliberately nothing here to hardcode. +# KMS is not real either: it will not release a key against a document it cannot +# trace to a Nitro root, so this image uses a static master key. +# +# What that leaves is everything except the hardware's signature, which is the +# part you cannot develop against anyway. +# --------------------------------------------------------------------------- +set -euo pipefail + +GUEST_WASM="" +PREFIX="${PREFIX:-dev}" +HTTPS_PORT="${HTTPS_PORT:-8443}" +WEBAUTHN_RP_ID="" +WEBAUTHN_ALLOWED_ORIGINS="" +GUEST_EGRESS_ORIGINS="" +BACKGROUND_TIMEOUT_SECS="" +GUEST_ENV="" +TLS_DOMAIN="" +ACME_STAGING="" +ACME_CONTACT="" +FCM_PROJECT="" +FCM_SERVICE_ACCOUNT="" +QEMU_MEMORY="${QEMU_MEMORY:-3G}" +STORE_BIND="" +PUBLISH_HOOK="" +PACK_DIR="" +BUNDLE="" + +usage() { + cat >&2 <. Repeatable. + An Android app claims android:apk-key-hash:. + --keep-store + keep the store — tenants, passkeys, everything guests wrote — in + target/qemu-nitro/-store, so the next start with the same name + resumes it instead of booting into genesis. A new guest is an + upgrade of the same store. The trust root and the attestation chain + are still new every boot, so clients re-read their pins. + --fresh with --keep-store, discard the kept store first. + --guest-egress ORIGIN + an origin the guest may send requests to, as http(s)://host[:port]. + Repeatable. None by default, and then the guest has no outbound + network. The host running this script is 192.168.127.254 from + inside the enclave. A different image, so a different PCR0. + --background-timeout SECS + how long one background task may run (the runtime's default is 30). + --guest-env NAME=VALUE + a variable for the guest, e.g. ASP_URL=http://192.168.127.254:7070. + Repeatable. Baked into the image, so measured by PCR0. + + For an emulator on a public host — test infrastructure, not a trust boundary: whoever runs the + host can read every tenant's data and sign attestation documents. See docs/DEV_ENCLAVE.md. + + --domain NAME + serve NAME with a certificate from Let's Encrypt, validated over + TLS-ALPN-01 on this host's --port, which therefore has to be 443 and + reachable from the internet, with NAME resolving here. No Pebble. + --acme-staging + Let's Encrypt's staging CA: prove the setup before spending the + production CA's rate limit. Its certificates are trusted by nothing. + --acme-contact EMAIL + the address Let's Encrypt sends expiry notices to. + --fcm-project ID --fcm-service-account FILE + real Firebase notifications, with that service account's JSON key, + instead of the stub. The key is baked into the image. + --memory SIZE + the enclave's memory, as QEMU's -m. Default $QEMU_MEMORY. + --store-bind ADDR + publish MinIO on ADDR only, e.g. 127.0.0.1 — its credentials are + the well-known defaults. + --publish-hook CMD + run CMD with this run's directory once the enclave is up: the trust + root is new every boot, so whatever pins it needs the new one. + --pack DIR + build the image and every host binary into DIR and stop. Image + options apply; run options do not. + --prebuilt DIR + run from a DIR --pack made, on a host with Docker, KVM and vsock but + no Nix, cargo or git. Image options are fixed by the pack and refused. +EOF + exit 2 +} + +# Each value is required rather than defaulted to empty, and `--name` is +# checked against a pattern. lib.sh derives RUNDIR from it and clears that +# directory before a run, so an empty or `..`-bearing name would have it delete +# a directory nobody asked it to — including the tools installed under +# target/qemu-nitro. +need() { [[ -n "${2:-}" ]] || { echo "$1 needs a value" >&2; usage; }; } + +while [[ $# -gt 0 ]]; do + case "$1" in + --guest) need "$1" "${2:-}"; GUEST_WASM="$2"; shift 2 ;; + --port) need "$1" "${2:-}"; HTTPS_PORT="$2"; shift 2 ;; + --name) need "$1" "${2:-}"; PREFIX="$2"; shift 2 ;; + --rp-id) need "$1" "${2:-}"; WEBAUTHN_RP_ID="$2"; shift 2 ;; + --keep-store) KEEP_STORE=1; shift ;; + --guest-egress) + need "$1" "${2:-}" + GUEST_EGRESS_ORIGINS="${GUEST_EGRESS_ORIGINS:+$GUEST_EGRESS_ORIGINS,}$2" + shift 2 ;; + --background-timeout) need "$1" "${2:-}"; BACKGROUND_TIMEOUT_SECS="$2"; shift 2 ;; + --guest-env) need "$1" "${2:-}"; GUEST_ENV="${GUEST_ENV:+$GUEST_ENV,}$2"; shift 2 ;; + --fresh) FRESH_STORE=1; shift ;; + --allowed-origin) + need "$1" "${2:-}" + WEBAUTHN_ALLOWED_ORIGINS="${WEBAUTHN_ALLOWED_ORIGINS:+$WEBAUTHN_ALLOWED_ORIGINS,}$2" + shift 2 ;; + --domain) need "$1" "${2:-}"; TLS_DOMAIN="$2"; shift 2 ;; + --acme-staging) ACME_STAGING=1; shift ;; + --acme-contact) need "$1" "${2:-}"; ACME_CONTACT="$2"; shift 2 ;; + --fcm-project) need "$1" "${2:-}"; FCM_PROJECT="$2"; shift 2 ;; + --fcm-service-account) need "$1" "${2:-}"; FCM_SERVICE_ACCOUNT="$2"; shift 2 ;; + --memory) need "$1" "${2:-}"; QEMU_MEMORY="$2"; shift 2 ;; + --store-bind) need "$1" "${2:-}"; STORE_BIND="$2"; shift 2 ;; + --publish-hook) need "$1" "${2:-}"; PUBLISH_HOOK="$2"; shift 2 ;; + --pack) need "$1" "${2:-}"; PACK_DIR="$2"; shift 2 ;; + --prebuilt) need "$1" "${2:-}"; BUNDLE="$2"; shift 2 ;; + -h|--help) usage ;; + *) echo "unknown argument: $1" >&2; usage ;; + esac +done + +[[ "$PREFIX" =~ ^[a-zA-Z0-9][a-zA-Z0-9_-]*$ ]] \ + || { echo "--name must be letters, digits, _ or -, and start with a letter or digit" >&2; exit 1; } +[[ "$HTTPS_PORT" =~ ^[0-9]+$ ]] && (( HTTPS_PORT > 0 && HTTPS_PORT < 65536 )) \ + || { echo "--port must be a port number" >&2; exit 1; } + +# An absolute path, resolved before lib.sh changes anything: the component is +# named relative to wherever the caller ran this from, which is very unlikely to +# be this directory. +if [[ -n "$GUEST_WASM" ]]; then + GUEST_WASM="$(readlink -f "$GUEST_WASM")" \ + || { echo "no such component: $GUEST_WASM" >&2; exit 1; } +fi + +# Both end up inside a Nix expression, so they are held to exactly the shapes they can take. +[[ -z "$WEBAUTHN_RP_ID" || "$WEBAUTHN_RP_ID" =~ ^[a-z0-9]([a-z0-9-]*[a-z0-9])?(\.[a-z0-9]([a-z0-9-]*[a-z0-9])?)+$ ]] \ + || { echo "--rp-id must be a domain name" >&2; exit 1; } +IFS=, read -ra origins <<<"$WEBAUTHN_ALLOWED_ORIGINS" +for o in "${origins[@]}"; do + [[ "$o" =~ ^(https://[a-z0-9.-]+(:[0-9]+)?|android:apk-key-hash:[A-Za-z0-9_-]{43})$ ]] \ + || { echo "--allowed-origin $o: expected https:// or android:apk-key-hash:<43 characters>" >&2; exit 1; } +done +IFS=, read -ra egress <<<"$GUEST_EGRESS_ORIGINS" +for o in "${egress[@]}"; do + [[ "$o" =~ ^https?://[a-z0-9.-]+(:[0-9]{1,5})?$ ]] \ + || { echo "--guest-egress $o: expected http(s)://host[:port]" >&2; exit 1; } +done +IFS=, read -ra guest_env <<<"$GUEST_ENV" +for kv in "${guest_env[@]}"; do + [[ "$kv" =~ ^[A-Z_][A-Z0-9_]*=[A-Za-z0-9:/._-]*$ ]] \ + || { echo "--guest-env $kv: expected NAME=VALUE (letters, digits and :/._- in the value)" >&2; exit 1; } + [[ "$kv" != S3FS_* && "$kv" != AWS_* ]] \ + || { echo "--guest-env $kv: S3FS_ and AWS_ variables are withheld from guests" >&2; exit 1; } +done +[[ -z "$BACKGROUND_TIMEOUT_SECS" || "$BACKGROUND_TIMEOUT_SECS" =~ ^[1-9][0-9]{0,5}$ ]] \ + || { echo "--background-timeout must be a number of seconds" >&2; exit 1; } +if [[ -n "$WEBAUTHN_ALLOWED_ORIGINS" && -z "$WEBAUTHN_RP_ID" ]]; then + echo "--allowed-origin needs --rp-id: an app's origin is only ever vouched for by its own domain" >&2 + exit 1 +fi + +[[ -z "$TLS_DOMAIN" || "$TLS_DOMAIN" =~ ^[a-z0-9]([a-z0-9-]*[a-z0-9])?(\.[a-z0-9]([a-z0-9-]*[a-z0-9])?)+$ ]] \ + || { echo "--domain must be a domain name" >&2; exit 1; } +[[ -z "$ACME_STAGING$ACME_CONTACT" || -n "$TLS_DOMAIN$BUNDLE" ]] \ + || { echo "--acme-staging and --acme-contact need --domain: Pebble takes neither" >&2; exit 1; } +[[ -z "$ACME_CONTACT" || "$ACME_CONTACT" =~ ^[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+$ ]] \ + || { echo "--acme-contact must be an email address" >&2; exit 1; } +if [[ -n "$FCM_PROJECT$FCM_SERVICE_ACCOUNT" ]]; then + [[ -n "$FCM_PROJECT" && -n "$FCM_SERVICE_ACCOUNT" ]] \ + || { echo "--fcm-project and --fcm-service-account go together" >&2; exit 1; } + [[ "$FCM_PROJECT" =~ ^[a-z0-9-]+$ ]] || { echo "--fcm-project must be a Firebase project id" >&2; exit 1; } + FCM_SERVICE_ACCOUNT="$(readlink -f "$FCM_SERVICE_ACCOUNT")" && [[ -f "$FCM_SERVICE_ACCOUNT" ]] \ + || { echo "no such service account file" >&2; exit 1; } + # A Nix path literal: no spaces or quotes to break out of the expression. + [[ "$FCM_SERVICE_ACCOUNT" =~ ^/[A-Za-z0-9/._-]+$ ]] \ + || { echo "--fcm-service-account path must be letters, digits and /._-" >&2; exit 1; } + jq -e '.type == "service_account" and .project_id != null' "$FCM_SERVICE_ACCOUNT" >/dev/null \ + || { echo "--fcm-service-account is not a service account key" >&2; exit 1; } +fi +[[ "$QEMU_MEMORY" =~ ^[1-9][0-9]*[MG]$ ]] || { echo "--memory must be like 1536M or 3G" >&2; exit 1; } +[[ -z "$STORE_BIND" || "$STORE_BIND" =~ ^[0-9.]+$ ]] || { echo "--store-bind must be an IPv4 address" >&2; exit 1; } +[[ -z "$PACK_DIR" || -z "$BUNDLE" ]] || { echo "--pack and --prebuilt are exclusive" >&2; exit 1; } +if [[ -n "$PACK_DIR" ]]; then + mkdir -p "$PACK_DIR" && PACK_DIR="$(readlink -f "$PACK_DIR")" +fi +if [[ -n "$BUNDLE" ]]; then + # The image is what it was packed as. An image option here would describe an enclave that is + # not the one about to boot, and the summary and the passkey client would believe it. + [[ -z "$WEBAUTHN_RP_ID$WEBAUTHN_ALLOWED_ORIGINS$GUEST_EGRESS_ORIGINS$BACKGROUND_TIMEOUT_SECS$GUEST_ENV$TLS_DOMAIN$ACME_STAGING$ACME_CONTACT$FCM_PROJECT" ]] \ + || { echo "--prebuilt fixes the image: pass image options to --pack instead" >&2; exit 1; } + BUNDLE="$(readlink -f "$BUNDLE")" && [[ -f "$BUNDLE/image.env" ]] \ + || { echo "--prebuilt $BUNDLE: not a bundle (no image.env)" >&2; exit 1; } + [[ -n "$GUEST_WASM" ]] || { echo "--prebuilt needs --guest: a bundle carries no guest" >&2; exit 1; } + # shellcheck source=/dev/null + source "$BUNDLE/image.env" +fi + +if [[ -n "${FRESH_STORE:-}" && -z "${KEEP_STORE:-}" ]]; then + echo "--fresh only means something with --keep-store: without it every start is fresh" >&2 + exit 1 +fi + +export GUEST_WASM PREFIX HTTPS_PORT WEBAUTHN_RP_ID WEBAUTHN_ALLOWED_ORIGINS KEEP_STORE FRESH_STORE \ + GUEST_EGRESS_ORIGINS BACKGROUND_TIMEOUT_SECS GUEST_ENV TLS_DOMAIN ACME_STAGING ACME_CONTACT \ + FCM_PROJECT FCM_SERVICE_ACCOUNT QEMU_MEMORY STORE_BIND PACK_DIR BUNDLE +# shellcheck source=lib.sh +source "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/lib.sh" + +if [[ -n "$PACK_DIR" ]]; then + enclave_pack + exit 0 +fi + +enclave_bring_up + +# The passkey the summary below is written against, so the command it prints +# works as printed. Registration is open — any passkey may enrol and gets its +# own tenant — so this is one identity among however many a client goes on to +# create, not a credential the enclave holds. +# +# `enrol`, not a signed request to some route: this serves whichever component +# was handed to it, and a guest of somebody else's has no route this script +# could know to ask for. Enrolment goes to the runtime and never reaches the +# guest, so it works against any of them. +say "enrolling a passkey, so there is something to try" +signed enrol >/dev/null || fail "the enclave refused to enrol a passkey" + +if [[ -n "$PUBLISH_HOOK" ]]; then + say "publishing this boot's pins" + # Not fatal: the enclave is up and serving either way, and a hook that failed says so itself. + $PUBLISH_HOOK "$RUNDIR" || echo "the publish hook failed; clients still hold the previous pins" >&2 +fi + +if [[ -n "${KEEP_STORE:-}" ]]; then + store_note="The store in $STORE_DIR is kept: it resumes, with its tenants and +their data, and the new guest is an upgrade of it. --fresh discards it." +else + store_note="That is a fresh start, not a reload: the store is rebuilt, so the +enclave boots into genesis rather than resuming. --keep-store keeps it." +fi + +if [[ -n "$FCM_PROJECT" ]]; then + wakes="Firebase project $FCM_PROJECT" +else + wakes="$FCM_RECORD (every notification the guest raised, as JSON)" +fi +if [[ -n "$TLS_DOMAIN" ]]; then + certificate_note="The certificate is from Let's Encrypt." + [[ -z "$ACME_STAGING" ]] || certificate_note="The certificate is from Let's Encrypt's staging CA, which nothing trusts." +else + certificate_note="The certificate is issued by Pebble, which no public root signs, so a client +also needs Pebble's root — or has to pin the certificate out of the attestation +document, which is what a real client should do anyway: + + --ca $RUNDIR/pebble-root.pem" +fi + +cat <${KEEP_STORE:+ --keep-store} + +Ctrl-C to stop everything. +EOF + +# The trap in lib.sh does the tearing down; this only has to stay alive for it. +# `wait` rather than a sleep loop, so Ctrl-C is handled at once. +while true; do + if ! docker inspect -f '{{.State.Running}}' "$PREFIX-qemu" 2>/dev/null | grep -q true; then + echo + echo "the enclave stopped. Last of its console:" >&2 + plain | tail -20 >&2 + exit 1 + fi + sleep 5 & + wait $! +done diff --git a/deploy/qemu-nitro/fcm-stub.py b/deploy/qemu-nitro/fcm-stub.py new file mode 100755 index 0000000..fd3d602 --- /dev/null +++ b/deploy/qemu-nitro/fcm-stub.py @@ -0,0 +1,73 @@ +#!/usr/bin/env python3 +"""A stand-in for Firebase, so the e2e can prove what the enclave sends. + +Google is not reachable from this harness and would not accept a made-up +registration token if it were. What matters is not that FCM delivered anything +— it is that the *runtime* built the right request: a data-only message, with +no `notification` object, carrying nothing but the labels the guest chose. + +So this answers the two endpoints the runtime calls and writes every message it +receives to a file the harness reads back: + + POST /token -> an access token + POST /v1/projects//messages:send -> recorded, then accepted + +Plaintext HTTP, which the runtime allows only because `--fcm-endpoint` was +pointed here; PCR0 records that the emulator image was built that way. A +production image has no such setting and talks to Google over TLS it validates +against roots compiled into the binary. +""" + +import json +import sys +from http.server import BaseHTTPRequestHandler, HTTPServer + +RECORD = sys.argv[1] if len(sys.argv) > 1 else "/tmp/fcm-messages.jsonl" + + +class Handler(BaseHTTPRequestHandler): + def do_POST(self): # noqa: N802 - the base class names it + length = int(self.headers.get("content-length", 0)) + body = self.rfile.read(length) + + if self.path.endswith("/token"): + # The runtime signs a real RS256 assertion to get here. Checking it + # would mean holding the public key and reimplementing Google; the + # signing itself is covered by a unit test that verifies against the + # key that produced it. + return self.reply(200, {"access_token": "stub-token", "expires_in": 3600}) + + if self.path.endswith("messages:send"): + try: + message = json.loads(body) + except json.JSONDecodeError: + return self.reply(400, {"error": {"status": "INVALID_ARGUMENT"}}) + with open(RECORD, "a", encoding="utf-8") as f: + f.write(json.dumps({ + "authorization": self.headers.get("authorization", ""), + "path": self.path, + "message": message.get("message", {}), + }) + "\n") + f.flush() + return self.reply(200, {"name": "projects/e2e/messages/stub"}) + + self.reply(404, {"error": {"status": "NOT_FOUND"}}) + + def reply(self, status, payload): + encoded = json.dumps(payload).encode() + self.send_response(status) + self.send_header("content-type", "application/json") + self.send_header("content-length", str(len(encoded))) + self.end_headers() + self.wfile.write(encoded) + + def log_message(self, *_args): + # The harness reads the record file; per-request noise on the console + # would bury the legs it is actually asserting. + pass + + +if __name__ == "__main__": + open(RECORD, "w", encoding="utf-8").close() + print(f"fcm-stub: listening on 9101, recording to {RECORD}", flush=True) + HTTPServer(("0.0.0.0", 9101), Handler).serve_forever() diff --git a/deploy/qemu-nitro/fcm/README.md b/deploy/qemu-nitro/fcm/README.md new file mode 100644 index 0000000..24f1dac --- /dev/null +++ b/deploy/qemu-nitro/fcm/README.md @@ -0,0 +1,11 @@ +A test service account for the end-to-end harness, and the stub it talks to. + +`service-account.json` is a throwaway RSA key with a `token_uri` pointing at +`fcm-stub.py` on the host rather than at Google. It is committed for the same +reason `../pebble/key.pem` is: the emulator image has to carry *something*, and +a fixture keeps the run deterministic. Nothing outside this directory reads it, +and the production image sets these values empty. + +It is minified on purpose. `nix/eif.nix` writes the image environment as +`KEY=value` lines joined by newlines, so a raw newline in the value would split +one setting into two. diff --git a/deploy/qemu-nitro/fcm/service-account.json b/deploy/qemu-nitro/fcm/service-account.json new file mode 100644 index 0000000..c97618f --- /dev/null +++ b/deploy/qemu-nitro/fcm/service-account.json @@ -0,0 +1 @@ +{"type":"service_account","project_id":"e2e","private_key_id":"qemu-e2e","private_key":"-----BEGIN PRIVATE KEY-----\nMIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQCvw2xq8t9LTXUH\ncyeYu7ZfDpvMoXMRXOtXTMczYrU6/L6XLFWacxeSdyCeEhzHLASDaR6UIQaVv4ic\nachgEJ8MCitwXf99VJ1h5IXFatdZJdGoI9FifWAfN/FKutFgPjO49FSdvJo8fG4u\nZr1hJ201LKrH4j9DvG1aSTk+5Q94f3p+ZHPJ2FkFjAicNFPprlVTm6H7Sct7VWYc\noRXkBMx4TxyhWm+AHZlFzKJ6ZIR8SLC+Ieu7+5YVsE3xvUBVMvrLcGpFl/ioMbTY\nHMTFooefdYbKaelUaJciv9lZnWKXhyvYPuYZwcC9sk09jeIWi9OlaBpDosPt6K70\nNQ2gWe5VAgMBAAECggEAHD/DTOQoteRm3xHnxxFCdEA3k7nOMff2fkNBj/V5Ldgh\n9M+kGY0OeJSreiRsmiltt0Y9qy6srYRJe2Q4F5KMUYXP6gE9l0HygqGVS3/KyVH+\nArFxDYyblqDp5+ojTT3qF7uzXt/JhVe1aMFMBlGtKHr7nuEy7FrcU4LJz91GcYX9\nGuj2oJJNwvBmoDR2w+4iJ7UZeYvpuqJ1qIs/YZVaE7dzFgyIbA5AR2s2/F+/OVDd\nePsqPDLOFHEyyhfYswMYaHaCMhixZkx6+y1i++lx0GIb2Thjb61TBWu/jHDsTB43\ntwI/V2NIcmYQmUrXODPwbW/+kFBlz+Cf5863dLzAKQKBgQDq0Lq/BmRzLY+u3UDo\nfYd27bz1rJif2DEmXonAWrfxTBJIoQrFKX9SsyzPJPbnKACuHB4jTVETHhjccKJ3\nFGHz1k986NaGo6iJ5UBin+Td3rIOzNrp781PZRBsLJW8P/siO14EZzFKlssFdkva\nZMnCkQlWOBb2/42zLvqMwxZaPQKBgQC/ntX4BiBWtJ0I6hHtjHEwMmZh0wMFBF8G\n8zpgUCySClnjaoK6kLA9hwy4zDPDOtIZOieHYs+Rg16EjsEMeCS52gmrpdpEE8ob\neCKQNViEPOXm3pJBsuIlVrodaWO4Oo1Pvhc8LZBTZ63qvXV0/pKzfnWAjS88Vwc0\n8nxUlWRd+QKBgB3tpq+sP+dSOkr+VkSLo1VsLbZeXkGZS4JpcEM9DM7LdFUfeYDx\nrhG7Vo28V1/VAGkwmkLDmv7FykNmc76bsXRjr1PrVVRpzZRtzMwFNyV0OdubDpfc\ngZ2J8xLmh9sriHWvfWcwQ98O4yd6EWbvi6up0rfThFHM9qGM7lA8mT+9AoGAZHUr\nC8p6bbpmkWPVXkpAlNn3XtW3QYwXHZeqRRADLdULZvRR8Okl3DvO6Zr0kCdoOh2I\n16tv0oOiq7ADeTwLVPwAEeLzWLlfPaNvy1aMP1eF19Fbr+HOOXEMRZsY0l6v8txf\nZgclIPS78tK8n0dPNZbYlzptRx8BAjsV/2oKolECgYA0sPm2WbpwEApgeMc+nXbP\ntbmPLZcEpXSm9Z3UdM0GnkY3DBGGRXyH1At3KlRIgBq4zN8ojNKn8cbSgKJ/a8aT\nLy6fCf18shOUq3jww1TZOa1TkUZFWNfdTjTIMXEzH/dvIlJtlJjWmStp+qrlBKYE\nGBs4U6iXoioFnUaCwPT65w==\n-----END PRIVATE KEY-----\n","client_email":"wake@e2e.iam.gserviceaccount.com","token_uri":"http://192.168.127.254:9101/token"} \ No newline at end of file diff --git a/deploy/qemu-nitro/heartbeat.py b/deploy/qemu-nitro/heartbeat.py new file mode 100755 index 0000000..82bd9ee --- /dev/null +++ b/deploy/qemu-nitro/heartbeat.py @@ -0,0 +1,40 @@ +#!/usr/bin/env python3 +"""Answer the enclave init's boot heartbeat. + +A real enclave's `init` proves it is alive to the parent instance: it connects +to the parent over vsock on port 9000, writes one byte (0xB7) and waits to read +the same byte back. Only then does it run the application. There is no timeout +and no fallback — with nobody listening, the enclave boots the kernel, prints +nothing further, and sits there. That silence looks exactly like a broken image, +which is why this tiny thing exists as its own file rather than an inline +one-liner: it is the first thing to check when a boot appears to hang. + +On real hardware the parent side is `nitro-cli`. Here the parent is this +script, listening on the host's own AF_VSOCK. It cannot be a Unix socket: +`vhost-device-vsock`'s `uds-path` backend only accepts packets addressed to the +host CID (2), and the enclave dials **CID 3**, the Nitro parent convention. It +drops the rest with `dropping packet for unknown cid: 3`. So the backend runs +with `forward-cid=1` instead, which turns the guest's connection into a real +host vsock connection to the loopback CID — and that needs `vsock_loopback` +loaded on the host. + +Usage: heartbeat.py +""" +import socket +import sys + +port = int(sys.argv[1]) + +srv = socket.socket(socket.AF_VSOCK, socket.SOCK_STREAM) +srv.bind((socket.VMADDR_CID_ANY, port)) +srv.listen(8) +print(f"heartbeat: listening on vsock port {port}", flush=True) + +while True: + conn, _ = srv.accept() + with conn: + data = conn.recv(1) + if not data: + continue + print(f"heartbeat: received 0x{data[0]:02x}, echoing", flush=True) + conn.sendall(data) diff --git a/deploy/qemu-nitro/lib.sh b/deploy/qemu-nitro/lib.sh new file mode 100644 index 0000000..0a7ad59 --- /dev/null +++ b/deploy/qemu-nitro/lib.sh @@ -0,0 +1,864 @@ +# Standing up the whole stack in an emulated enclave. +# +# Sourced, never run. Two things need this and they need exactly the same +# thing: `run-e2e.sh`, which brings the stack up and then asserts against it, +# and `dev-enclave.sh`, which brings it up and leaves it running for somebody to +# develop a client against. Every difference between them belongs after the +# enclave is serving, so everything before that lives here. +# +# host QEMU enclave +# ──── ──────────── +# MinIO :9000 ◀── gvproxy ──192.168.127.254── s3fs mount, and the guest +# gvproxy --listen vsock://:1024 ────────────▶ gvforwarder → tap0 .2 +# expose :8443 → 192.168.127.2:443 ──▶ rustls :443 +# vhost-device-vsock --forward-cid 1 +# heartbeat.py :9000 ───────────────────────▶ init's boot heartbeat +# Pebble :14000 ◀──────────────────────────── ACME order, TLS-ALPN-01 +# +# --------------------------------------------------------------------------- +# What a caller sets before sourcing, all optional: +# +# PREFIX names the run: $WORK/$PREFIX, and every container and docker +# label. Default `e2e`. It keeps two runs' files and containers +# apart; it does not let them run at the same time, because +# MinIO's :9000, the FCM stub's :9101 and the enclave's vsock CID +# are fixed — the last two by the image, which dials them. +# HTTPS_PORT host port gvproxy forwards to the enclave's :443, and the +# port Pebble validates against. Default 8443. +# GUEST_WASM a component to serve instead of the one Nix builds. This is +# what makes the emulator useful to somebody else's guest. +# WITH_SUBSTITUTE also stage a second, altered guest, for a caller that +# wants to show what happens when the object is replaced. +# TIMEOUT seconds for each wait. Default 240. +# QEMU_MEMORY the enclave's memory, as QEMU's -m. Default 3G. +# STORE_BIND the address MinIO is published on. Default all of them. +# TLS_DOMAIN serve this name with a certificate from Let's Encrypt instead +# of enclave.test from Pebble — for an emulator on a public host. +# ACME_STAGING picks the staging CA, ACME_CONTACT its contact. +# FCM_PROJECT / FCM_SERVICE_ACCOUNT +# real Firebase notifications instead of the stub. +# PACK_DIR build everything a run needs into this directory and stop. +# BUNDLE run from a directory PACK_DIR made: no Nix, no cargo, no git +# on this machine — see `enclave_pack`. +# +# What it gets back, after `enclave_bring_up`: +# +# EXPECTED_PCR0 / EXPECTED_PCR16 what the builds said the measurements are +# TRUST_ROOT DER the enclave's documents chain to +# ATTEST / PASSKEY client binaries, built here +# PASSKEY_ARGS the trust flags both clients need +# signed / signed2 two identities, each with its own passkey +# CONSOLE, RUNDIR, plain, fail, say, nonce, boot_enclave, upload_guest, +# wait_for_serving, wait_for_https +# --------------------------------------------------------------------------- + +REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +WORK="${WORK:-$REPO/target/qemu-nitro}" +PREFIX="${PREFIX:-e2e}" +RUNDIR="${RUNDIR:-$WORK/$PREFIX}" + +# The store, when it is kept across runs: MinIO's data, and every Pebble root +# and intermediate that has issued a certificate the store may still hold. +# Beside RUNDIR rather than in it, because RUNDIR is cleared on every start. +STORE_DIR="$WORK/$PREFIX-store" +KEEP_STORE="${KEEP_STORE:-}" +FRESH_STORE="${FRESH_STORE:-}" +IMAGE="${QEMU_IMAGE:-s3fs-qemu-nitro:latest}" +TIMEOUT="${TIMEOUT:-240}" +QEMU_MEMORY="${QEMU_MEMORY:-3G}" +STORE_BIND="${STORE_BIND:-}" +TLS_DOMAIN="${TLS_DOMAIN:-}" +ACME_STAGING="${ACME_STAGING:-}" +ACME_CONTACT="${ACME_CONTACT:-}" +FCM_PROJECT="${FCM_PROJECT:-}" +FCM_SERVICE_ACCOUNT="${FCM_SERVICE_ACCOUNT:-}" +PACK_DIR="${PACK_DIR:-}" +BUNDLE="${BUNDLE:-}" + +LETS_ENCRYPT_DIRECTORY=https://acme-v02.api.letsencrypt.org/directory +LETS_ENCRYPT_STAGING_DIRECTORY=https://acme-staging-v02.api.letsencrypt.org/directory +CONSOLE="$RUNDIR/console.log" + +# Host-side port that gvproxy forwards into the enclave's :443. Pebble also +# validates the TLS-ALPN-01 challenge against it, so it is both the service +# port and the challenge port — as it is in production, where both are 443. +HTTPS_PORT="${HTTPS_PORT:-8443}" + +# The ACME test CA. Pinned by digest: this image supplies the issuance path the +# harness claims to exercise, and "latest" would let that change under us. +PEBBLE_IMAGE="${PEBBLE_IMAGE:-ghcr.io/letsencrypt/pebble@sha256:ddf230642b1a584f519f32e347de1b05a6e4c1f6c35c1863b33effeab5f78199}" + +# Every container this run creates carries it, so cleanup can find them without +# a hard-coded list that the next container to be added would not appear in. +LABEL="enclave-harness=$PREFIX" + +# These do *not* follow PREFIX, and must not. They are baked into the emulator +# image as S3FS_BUCKET and S3FS_ROOTS_BUCKET (flake.nix, `eif-qemu`), which +# means they are covered by PCR0 — an enclave image is its configuration as much +# as its code. A run that named its buckets after itself would stand up a store +# the enclave does not look in, and the enclave would fail at mount with +# "connecting to the data bucket: not found". +DATA_BUCKET=e2e-data +ROOTS_BUCKET=e2e-roots + +say() { printf '\n== %s ==\n' "$*"; } + +# --------------------------------------------------------------------------- +# Preconditions, each with the fix rather than just the symptom. +# --------------------------------------------------------------------------- +enclave_preflight() { + if [[ -n "$BUNDLE" ]]; then + enclave_preflight_host + return + fi + command -v nix >/dev/null || { + echo "nix is not on PATH; see deploy/nix/README.md" >&2; exit 1; } + + # Flakes copy only what git tracks, so an untracked file is invisible to the + # build no matter that it is right there on disk. The failure lands deep + # inside the EIF derivation as a bare "cp: cannot stat", naming a store path + # that does not contain it — true, and useless. Checked here instead. + local f + for f in pebble/ca.pem pebble/cert.pem pebble/key.pem fcm/service-account.json; do + git -C "$REPO" ls-files --error-unmatch "deploy/qemu-nitro/$f" >/dev/null 2>&1 || { + echo "deploy/qemu-nitro/$f is not tracked by git, so nix cannot see it." >&2 + echo " git add deploy/qemu-nitro/" >&2 + exit 1 + } + done + + # And the general case, which cost twenty minutes to learn: a *source* file + # that git does not track is invisible to the build no matter that the + # workspace compiles here. A new module declared in lib.rs and never added + # fails as `file not found for module`, from a sandbox, after everything + # before it has been rebuilt. Caught in a second instead. + # + # Scoped to the trees that feed the image, and to the extensions whose + # absence breaks resolution rather than merely losing a file: a stray + # untracked note somewhere should not stop a run, and an untracked module, + # crate manifest, WIT package or flake input always should. + local untracked + untracked="$(git -C "$REPO" ls-files --others --exclude-standard \ + -- 'runtime/**/*.rs' 'runtime/**/*.toml' \ + 'crates/**/*.rs' 'crates/**/*.toml' \ + 'examples/**/*.rs' 'examples/**/*.toml' \ + '**/*.wit' '**/*.nix' 2>/dev/null || true)" + if [[ -n "$untracked" ]]; then + echo "these source files are not tracked by git, so nix will not see them:" >&2 + echo "$untracked" | sed 's/^/ /' >&2 + echo " git add , or remove them" >&2 + exit 1 + fi + VSOCK_BIN="$WORK/tools/bin/vhost-device-vsock" + [[ -x "$VSOCK_BIN" ]] || { + echo "missing $VSOCK_BIN (cargo install vhost-device-vsock --root $WORK/tools)" >&2 + exit 1 + } + enclave_preflight_host +} + +# What running needs, as opposed to building. A bundle is run on a machine that has none of the +# build's tools, so this is all it checks. +enclave_preflight_host() { + if [[ -n "$BUNDLE" ]]; then + [[ -f "$BUNDLE/eif/s3fs-qemu.eif" && -x "$BUNDLE/bin/gvproxy" ]] \ + || { echo "$BUNDLE is not a bundle: pack one with dev-enclave.sh --pack" >&2; exit 1; } + VSOCK_BIN="$BUNDLE/bin/vhost-device-vsock" + fi + # Packing only builds, so nothing it produces needs this machine to be able to run it. + if [[ -z "$PACK_DIR" ]]; then + [[ -e /dev/kvm ]] || { echo "no /dev/kvm — the nitro-enclave machine needs KVM" >&2; exit 1; } + [[ -e /dev/vsock ]] || { + echo "no /dev/vsock — the host needs: sudo modprobe vsock_loopback" >&2 + exit 1 + } + docker image inspect "$IMAGE" >/dev/null 2>&1 || { + echo "missing QEMU image $IMAGE — build it:" >&2 + echo " docker build -t $IMAGE deploy/qemu-nitro" >&2 + exit 1 + } + fi + + # Checked immediately before the `rm -rf`, because RUNDIR is overridable and + # this is the one line here that destroys something. `$WORK/` with an empty + # PREFIX, or an exported RUNDIR pointing anywhere else, would take the tools + # installed under target/qemu-nitro with it — or worse. + case "$RUNDIR" in + "$WORK"/?*) ;; + *) + echo "refusing to clear $RUNDIR: it must be a directory under $WORK" >&2 + exit 1 + ;; + esac + [[ "$RUNDIR" != *..* ]] || { echo "refusing to clear $RUNDIR: it contains .." >&2; exit 1; } + + rm -rf "$RUNDIR"; mkdir -p "$RUNDIR" +} + +pids=() +cleanup() { + # `kill 0` signals the whole process group, this script included, so a stray + # 0 in this array ends the harness by SIGTERM instead of letting it exit — + # which is how a run that had already printed PASS still reported 143. + for p in "${pids[@]:-}"; do + if [[ "$p" =~ ^[0-9]+$ ]] && (( p > 0 )); then + kill "$p" 2>/dev/null || true + fi + done + # `docker logs -f` never ends on its own, so the job table has to be killed + # rather than waited on. + local remaining + remaining="$(jobs -p 2>/dev/null || true)" + if [[ -n "$remaining" ]]; then + # shellcheck disable=SC2086 + kill $remaining 2>/dev/null || true + fi + # By label, not by name. A name list has to be kept in step with every + # container the harness learns to start, and the one that gets forgotten is + # the one left running. + local stragglers + stragglers="$(docker ps -aq --filter "label=$LABEL" 2>/dev/null || true)" + if [[ -n "$stragglers" ]]; then + # shellcheck disable=SC2086 + docker rm -f $stragglers >/dev/null 2>&1 || true + fi + return 0 +} + +# --------------------------------------------------------------------------- +# The image, the guest, and the measurements they claim. +# --------------------------------------------------------------------------- + +# nix-portable keeps its store outside /nix except inside its own namespace, so +# `result/` may not resolve. Ask nix for the path and translate it if the +# symlink is dangling — on a machine with a normal Nix install the first branch +# is always the one taken. +resolve_out_link() { + local link="$1" dir + dir="$(readlink -f "$link" 2>/dev/null || true)" + if [[ ! -d "$dir" ]]; then + dir="$HOME/.nix-portable$(readlink "$link")" + fi + [[ -d "$dir" ]] || { echo "cannot resolve $link" >&2; exit 1; } + printf '%s' "$dir" +} + +# `| tail -3` on the build, which is what this replaces, kept the last three +# lines of a nix log — which on failure are the derivation summary and not the +# compiler error forty lines above it. The log is kept whole and only the tail +# shown, so a failure can say what actually went wrong. +nix_build() { + local attr="$1" link="$2" log="$RUNDIR/nix-${1//[^a-zA-Z0-9]/-}.log" + if ! nix build "$REPO#$attr" --out-link "$link" --print-build-logs > "$log" 2>&1; then + echo "nix build of $attr failed:" >&2 + grep -E 'error(\[|:)' "$log" | head -20 >&2 || true + echo " full log: $log" >&2 + exit 1 + fi + tail -3 "$log" +} + +nix_expr() { + local what="$1" link="$2" expr="$3" log="$RUNDIR/nix-eif-qemu.log" + if ! nix build --impure --expr "$expr" --out-link "$link" --print-build-logs > "$log" 2>&1; then + echo "nix build of $what failed:" >&2 + grep -E 'error(\[|:)' "$log" | head -20 >&2 || true + echo " full log: $log" >&2 + exit 1 + fi + tail -3 "$log" +} + +enclave_build_image() { + if [[ -n "$BUNDLE" ]]; then + EIF_DIR="$BUNDLE/eif" + EIF="$EIF_DIR/s3fs-qemu.eif" + EXPECTED_PCR0="$(jq -r .PCR0 "$EIF_DIR/pcr.json")" + echo "EIF $EIF (prebuilt)" + echo "PCR0 $EXPECTED_PCR0" + return + fi + say "building the enclave image" + if [[ -n "${WEBAUTHN_RP_ID:-}${GUEST_EGRESS_ORIGINS:-}${BACKGROUND_TIMEOUT_SECS:-}${GUEST_ENV:-}${TLS_DOMAIN}${FCM_PROJECT}" ]]; then + # The same image configured differently. Not a package — flake outputs take no arguments — + # so the flake's `lib.eifQemu` is called directly, which reading the flake by path needs + # --impure for. dev-enclave.sh has already held every value to a shape that cannot break + # out of the string. + nix_list() { + local out="" o + IFS=, read -ra items <<<"$1" + for o in "${items[@]}"; do out+=" \"$o\""; done + printf '[%s ]' "$out" + } + local args="" + [[ -n "${WEBAUTHN_RP_ID:-}" ]] && args+=" rpId = \"$WEBAUTHN_RP_ID\";" + [[ -n "${WEBAUTHN_ALLOWED_ORIGINS:-}" ]] \ + && args+=" allowedOrigins = $(nix_list "$WEBAUTHN_ALLOWED_ORIGINS");" + [[ -n "${GUEST_EGRESS_ORIGINS:-}" ]] \ + && args+=" guestEgressOrigins = $(nix_list "$GUEST_EGRESS_ORIGINS");" + [[ -n "${BACKGROUND_TIMEOUT_SECS:-}" ]] \ + && args+=" backgroundTimeoutSecs = $BACKGROUND_TIMEOUT_SECS;" + if [[ -n "${GUEST_ENV:-}" ]]; then + local env_attrs="" kv + IFS=, read -ra kvs <<<"$GUEST_ENV" + for kv in "${kvs[@]}"; do env_attrs+=" ${kv%%=*} = \"${kv#*=}\";"; done + args+=" guestEnv = {$env_attrs };" + fi + if [[ -n "$TLS_DOMAIN" ]]; then + local directory="$LETS_ENCRYPT_DIRECTORY" + [[ -n "$ACME_STAGING" ]] && directory="$LETS_ENCRYPT_STAGING_DIRECTORY" + args+=" tlsDomains = [ \"$TLS_DOMAIN\" ]; acmeDirectory = \"$directory\"; acmeCa = null;" + # A bare address: the runtime adds the mailto: scheme itself. + [[ -n "$ACME_CONTACT" ]] && args+=" acmeContacts = [ \"$ACME_CONTACT\" ];" + fi + if [[ -n "$FCM_PROJECT" ]]; then + args+=" fcm = { projectId = \"$FCM_PROJECT\"; serviceAccountFile = $FCM_SERVICE_ACCOUNT; };" + fi + nix_expr "eif-qemu (${args# })" "$RUNDIR/eif" \ + "((builtins.getFlake \"git+file://$REPO\").lib.\${builtins.currentSystem}.eifQemu {$args })" + else + nix_build eif-qemu "$RUNDIR/eif" + fi + EIF_DIR="$(resolve_out_link "$RUNDIR/eif")" + EIF="$EIF_DIR/s3fs-qemu.eif" + EXPECTED_PCR0="$(jq -r .PCR0 "$EIF_DIR/pcr.json")" + echo "EIF $EIF ($(du -h "$EIF" | cut -f1))" + echo "PCR0 $EXPECTED_PCR0" +} + +# The guest is not in the image. It is uploaded to the store and the enclave +# measures what it fetches into PCR16 — so the build says what PCR16 should be, +# the same way `pcr.json` says what PCR0 should be. +enclave_build_guest() { + mkdir -p "$RUNDIR/guests" + if [[ -n "${GUEST_WASM:-}" ]]; then + # Somebody else's component. Nothing claims what it measures to, so the + # measurement comes from the verifier below rather than from a build. + say "staging the guest" + [[ -f "$GUEST_WASM" ]] || { echo "no such component: $GUEST_WASM" >&2; exit 1; } + install -m 0644 "$GUEST_WASM" "$RUNDIR/guests/guest.wasm" + echo "guest $GUEST_WASM ($(du -h "$GUEST_WASM" | cut -f1))" + else + say "building the guest release" + nix_build guest-release "$RUNDIR/guest" + GUEST_DIR="$(resolve_out_link "$RUNDIR/guest")" + EXPECTED_PCR16="$(jq -r .PCR16 "$GUEST_DIR/guest-pcr16.json")" + install -m 0644 "$GUEST_DIR/guest.wasm" "$RUNDIR/guests/guest.wasm" + echo "PCR16 $EXPECTED_PCR16" + fi + + if [[ -n "${WITH_SUBSTITUTE:-}" ]]; then + # A second guest differing from the first only by an appended custom + # section: a valid component with a different hash, which is all a + # substituted object needs to be. + install -m 0644 "$RUNDIR/guests/guest.wasm" "$RUNDIR/guests/substitute.wasm" + printf '\x00\x0b\x0asubstitute' >> "$RUNDIR/guests/substitute.wasm" + fi +} + +# Built with the host's cargo rather than Nix. These are client-side tools that +# nothing attests, so they gain nothing from a reproducible build — and a +# Nix-built dynamic binary links against a glibc in the Nix store, which will +# not run on a machine that has no such store path. Only what goes *inside* the +# enclave has to come from Nix. +enclave_build_clients() { + if [[ -n "$BUNDLE" ]]; then + ATTEST="$BUNDLE/bin/nitro-attest" + PASSKEY="$BUNDLE/bin/passkey-client" + EXPECTED_PCR16="$("$ATTEST" --measure "$RUNDIR/guests/guest.wasm" | jq -r .PCR16)" + echo "PCR16 $EXPECTED_PCR16" + return + fi + say "building the verifier" + ( cd "$REPO" && cargo build --release -p nitro-attestation --features cli ) 2>&1 | tail -2 + ATTEST="$REPO/target/release/nitro-attest" + [[ -x "$ATTEST" ]] || { echo "nitro-attest did not build" >&2; exit 1; } + + if [[ -n "$PACK_DIR" ]]; then + : # No guest in a bundle: the run names its own, and measures it then. + elif [[ -n "${EXPECTED_PCR16:-}" ]]; then + # The release build measured the guest with a Nix-built nitro-attest; + # this one was built here. They must agree, or a key policy and a client + # would pin different numbers for the same guest. + [[ "$("$ATTEST" --measure "$RUNDIR/guests/guest.wasm" | jq -r .PCR16)" == "$EXPECTED_PCR16" ]] \ + || { echo "the release build and this verifier disagree about the guest's PCR16" >&2; exit 1; } + else + EXPECTED_PCR16="$("$ATTEST" --measure "$RUNDIR/guests/guest.wasm" | jq -r .PCR16)" + echo "PCR16 $EXPECTED_PCR16" + fi + + if [[ -n "${WITH_SUBSTITUTE:-}" ]]; then + SUBSTITUTE_PCR16="$("$ATTEST" --measure "$RUNDIR/guests/substitute.wasm" | jq -r .PCR16)" + [[ "$SUBSTITUTE_PCR16" != "$EXPECTED_PCR16" ]] \ + || { echo "the substitute guest measures the same as the real one" >&2; exit 1; } + fi + + # The client half of the WebAuthn gate. Nothing reaches the guest without a + # fresh assertion bound to that exact request, and a shell script cannot + # sign one — so the harness drives the gate with a software passkey. + ( cd "$REPO" && cargo build --release -p enclave-runtime --features testing \ + --bin passkey-client ) 2>&1 | tail -2 + PASSKEY="$REPO/target/release/passkey-client" + [[ -x "$PASSKEY" ]] || { echo "passkey-client did not build" >&2; exit 1; } +} + +# Two identities, each with its own passkey file, so a caller can show that one +# tenant cannot see another's data. +# +# `PASSKEY_ARGS` is empty until the enclave has booted and reported the root its +# documents chain to — see `enclave_trust_root`. Both functions read it at call +# time for that reason, and because each reboot mints a new chain. +signed() { "$PASSKEY" --url "https://127.0.0.1:$HTTPS_PORT" --state "$RUNDIR/alice.json" "${PASSKEY_ARGS[@]}" "$@"; } +signed2() { "$PASSKEY" --url "https://127.0.0.1:$HTTPS_PORT" --state "$RUNDIR/bob.json" "${PASSKEY_ARGS[@]}" "$@"; } + +# --------------------------------------------------------------------------- +# The store the enclave will mount, and the guest it will fetch. +# --------------------------------------------------------------------------- +enclave_start_store() { + local data_dir="" + if [[ -n "$KEEP_STORE" ]]; then + if [[ -n "$FRESH_STORE" && -d "$STORE_DIR" ]]; then + # The same guard RUNDIR has: this is the other line that destroys. + case "$STORE_DIR" in + "$WORK"/?*-store) ;; + *) echo "refusing to clear $STORE_DIR: not a store under $WORK" >&2; exit 1 ;; + esac + say "clearing the kept store" + rm -rf "$STORE_DIR" + fi + mkdir -p "$STORE_DIR" + # A stable name for this store, for clients that keep state per store: + # the trust root changes every boot, so it cannot be that. + [[ -s "$STORE_DIR/id" ]] || openssl rand -hex 16 > "$STORE_DIR/id" + data_dir="$STORE_DIR/minio" + say "starting MinIO on the kept store ($STORE_DIR)" + else + say "starting MinIO" + fi + # The same recipe the SQLite CI job used. It lived in run-e2e.sh in full + # until both copies had to agree with nothing making them agree. + MINIO_LABEL="$LABEL" MINIO_BIND="$STORE_BIND" "$REPO/scripts/minio-up.sh" \ + "$PREFIX-minio" 9000 "$DATA_BUCKET" "$ROOTS_BUCKET" $data_dir + upload_guest guest.wasm + echo "MinIO ready with $DATA_BUCKET and $ROOTS_BUCKET, and the guest at $ROOTS_BUCKET/guest/guest.wasm" +} + +# Where the emulator image looks: `deployment.guestObject` in the roots bucket. +upload_guest() { + local image + image="$("$REPO/scripts/minio-image.sh")" + docker run --rm --network host --label "$LABEL" -v "$RUNDIR/guests:/guests:ro" \ + --entrypoint sh "$image" -c " + mc alias set m http://127.0.0.1:9000 minioadmin minioadmin >/dev/null + mc cp /guests/$1 m/$ROOTS_BUCKET/guest/guest.wasm >/dev/null" >/dev/null \ + || { echo "could not upload $1 to the store" >&2; exit 1; } +} + +# --------------------------------------------------------------------------- +# The parent side: heartbeat, notifications, vsock transport, and the network. +# --------------------------------------------------------------------------- +enclave_start_parent() { + say "starting the parent-side services" + + # init writes 0xB7 to the parent on port 9000 and waits. Unanswered, the + # kernel boots and then nothing happens at all. + python3 "$REPO/deploy/qemu-nitro/heartbeat.py" 9000 > "$RUNDIR/heartbeat.log" 2>&1 & + pids+=($!) + + # Firebase is not reachable from here and would refuse an invented + # registration token if it were. The stub answers the two endpoints the + # runtime calls and records every message, so a caller can assert on what + # the runtime *sent*. + # + # Not with real notifications: the image then talks to Google, and has no stub to dial. + FCM_RECORD="$RUNDIR/fcm-messages.jsonl" + if [[ -z "$FCM_PROJECT" ]]; then + python3 "$REPO/deploy/qemu-nitro/fcm-stub.py" "$FCM_RECORD" > "$RUNDIR/fcm-stub.log" 2>&1 & + pids+=($!) + for _ in $(seq 50); do + curl -sf -o /dev/null -X POST --data '{}' http://127.0.0.1:9101/token && break + sleep 0.1 + done + curl -sf -o /dev/null -X POST --data '{}' http://127.0.0.1:9101/token \ + || { echo "the FCM stub never answered:" >&2; cat "$RUNDIR/fcm-stub.log" >&2; exit 1; } + fi + + # forward-cid 1 turns the guest's vsock connections into host vsock loopback + # connections, which is the only arrangement that reaches a listener for the + # CID 3 the enclave dials. + RUST_LOG="${VSOCK_LOG:-info}" "$VSOCK_BIN" \ + --guest-cid 4 \ + --socket "$RUNDIR/vhost.socket" \ + --forward-cid 1 \ + > "$RUNDIR/vsock.log" 2>&1 & + pids+=($!) + + for _ in $(seq 50); do [[ -S "$RUNDIR/vhost.socket" ]] && break; sleep 0.1; done + [[ -S "$RUNDIR/vhost.socket" ]] \ + || { echo "vhost-device-vsock never came up:" >&2; cat "$RUNDIR/vsock.log" >&2; exit 1; } + + # The enclave's only route to anything. Without it the runtime waits for the + # gateway and then says so. + # + # The same pin as the gvforwarder inside the image, and built static for the + # same reason: it has to run here, and later on Amazon Linux, neither of + # which has a Nix store. + if [[ -n "$BUNDLE" ]]; then + GV_DIR="$BUNDLE" + else + nix_build gvproxy "$RUNDIR/gvproxy" >/dev/null + GV_DIR="$(resolve_out_link "$RUNDIR/gvproxy")" + fi + + "$GV_DIR/bin/gvproxy" \ + --listen "vsock://:1024" \ + --listen "unix://$RUNDIR/network.sock" \ + > "$RUNDIR/gvproxy.log" 2>&1 & + pids+=($!) + for _ in $(seq 50); do [[ -S "$RUNDIR/network.sock" ]] && break; sleep 0.1; done + [[ -S "$RUNDIR/network.sock" ]] \ + || { echo "gvproxy never opened its API socket:" >&2; cat "$RUNDIR/gvproxy.log" >&2; exit 1; } + echo "gvproxy listening on host vsock port 1024" +} + +# The CA. Pebble is a real RFC 8555 server, so the enclave runs the same +# issuance path it runs against Let's Encrypt: directory, account, order, +# TLS-ALPN-01 challenge, finalize, and a certificate sealed into the cache. +# +# `--network host` because two things must reach it: the enclave, which dials +# gvproxy's host address 192.168.127.254:14000, and this script on loopback. +# `--add-host` is what makes the challenge work — Pebble resolves the identifier +# `enclave.test` to the loopback address where gvproxy forwards :443 into the +# enclave, so validation arrives on the port the service uses. +# +# Two knobs are turned off because they exist to make clients prove they retry, +# and a flaky CA here would read as a flaky runtime: PEBBLE_VA_NOSLEEP skips a +# random pre-validation delay, PEBBLE_WFE_NONCEREJECT the deliberate 5% bad +# nonce. rustls-acme handles both; this harness is not the place to find out. +enclave_start_ca() { + say "starting Pebble, the ACME test CA" + docker rm -f "$PREFIX-pebble" >/dev/null 2>&1 || true + cat > "$RUNDIR/pebble-config.json" </dev/null \ + || { echo "Pebble did not start" >&2; exit 1; } + + for _ in $(seq "$TIMEOUT"); do + curl -sk --max-time 2 "https://127.0.0.1:14000/dir" >/dev/null 2>&1 && break + sleep 1 + done + curl -sk --max-time 5 "https://127.0.0.1:14000/dir" >/dev/null 2>&1 \ + || { echo "Pebble never answered:" >&2; docker logs "$PREFIX-pebble" 2>&1 | tail -20 >&2; exit 1; } + echo "Pebble serving its directory on :14000, validating :$HTTPS_PORT" +} + +# Before the enclave boots, not after. The enclave starts its ACME order as soon +# as it has a network, and the challenge is a connection *inbound* to :443 — so +# if this forward does not exist yet, the first order fails and the harness +# waits out a retry backoff for no reason. +enclave_expose_https() { + say "forwarding :$HTTPS_PORT into the enclave" + curl -sf --unix-socket "$RUNDIR/network.sock" \ + http://localhost/services/forwarder/expose \ + -X POST -H 'Content-Type: application/json' \ + -d "{\"local\":\":${HTTPS_PORT}\",\"remote\":\"192.168.127.2:443\"}" \ + || { echo "gvproxy refused to forward :$HTTPS_PORT" >&2; exit 1; } +} + +# --------------------------------------------------------------------------- +# Boot. +# --------------------------------------------------------------------------- + +# /dev/kvm is handed to the container, but qemu still picks its own accelerator +# unless told; a silent fall back to TCG leaves the guest kernel unable to +# calibrate its TSC and the boot stalls there past any timeout. That is what +# made this harness fail roughly half its runs while the runtime under test was +# fine, so the accelerator is named rather than hoped for. +boot_enclave() { + docker run --rm -d --name "$1" --label "$LABEL" \ + --device /dev/kvm \ + --network none \ + -v "$EIF_DIR:/eif:ro" \ + -v "$RUNDIR:/run/vsock" \ + "$IMAGE" \ + qemu-system-x86_64 \ + -M nitro-enclave,vsock=chr0,id="$PREFIX" \ + -accel kvm -cpu host \ + -kernel /eif/s3fs-qemu.eif \ + -chardev socket,id=chr0,path=/run/vsock/vhost.socket \ + -m "$QEMU_MEMORY" -smp 2 -nographic -no-reboot >/dev/null + docker logs -f "$1" > "$2" 2>&1 & +} + +# The runtime colourises its logs, so escape sequences land between a field name +# and its value — `addr[0m[2m=` — and a grep for "addr=" simply never +# matches. Everything that reads a console goes through this. +# Read through process substitution, never `plain | grep -q`. `grep -q` exits on +# its first match and closes the pipe; the `sed` upstream then dies of SIGPIPE, +# and under `set -o pipefail` that makes the whole pipeline report failure even +# though the pattern matched. It bites in proportion to how much of the file is +# left unread, so a pattern early in this console fails while a later one passes +# — which is exactly how it was found. +plain_of() { sed -e 's/\x1b\[[0-9;]*[a-zA-Z]//g' "$1" 2>/dev/null; } +plain() { plain_of "$CONSOLE"; } + +fail() { + echo + echo "FAIL: $*" >&2 + echo "--- console (last 40) ---" >&2; plain | tail -40 >&2 + echo "--- gvproxy ---" >&2; tail -10 "$RUNDIR/gvproxy.log" >&2 + echo "--- heartbeat ---" >&2; cat "$RUNDIR/heartbeat.log" >&2 + exit 1 +} + +nonce() { openssl rand 20 | basenc --base64url | tr -d '='; } + +# The enclave takes its address by DHCP from gvproxy, fetches and measures its +# guest, then mounts over the network before it listens, so this waits on the +# whole chain rather than on the boot alone. +wait_for_serving() { + local ready="" + for _ in $(seq "$TIMEOUT"); do + # "serving " with the bound address, not the earlier "serving guest" — + # that one is logged before the listener exists and racing it produces + # a connection-refused that looks like a networking fault. + if grep -qE "serving +addr=" <(plain); then ready=1; break; fi + grep -qE "Kernel panic|failed to start the guest" <(plain) && fail "the enclave died during boot" + sleep 1 + done + [[ -n "$ready" ]] || fail "the enclave never started serving within ${TIMEOUT}s" +} + +# Capture the root this boot's attestation documents chain to, and point the +# clients at it. +# +# The emulator image sets `S3FS_COSIGN_ATTESTATIONS`, so the runtime re-signs +# what QEMU's NSM produces — QEMU does not sign at all, and a client meeting an +# unsigned document has to skip the signature, the chain and the validity +# windows, which is most of what a client does. The contents stay the device's; +# only the envelope is added. It says nothing about *who* produced a document, +# because the key is minted inside an image whoever runs it controls. What it +# buys is that the client code being developed here is the code that will run +# against hardware, rather than a relaxed variant of it. +# +# The root is minted fresh at every boot and reported on the console, which is +# why this is called again after each reboot rather than once. +enclave_trust_root() { + local console="${1:-$CONSOLE}" + local b64="" + for _ in $(seq "$TIMEOUT"); do + # `|| true` is load-bearing under `set -o pipefail`. Before the line has + # been written grep matches nothing and exits 1; once it has, `head` + # closes the pipe and grep exits 141 instead. Either takes the whole + # script down inside this assignment, and *silently* — a failed + # assignment prints nothing, so the run ends with the last thing it said + # being the step header. That is exactly how this was found. + b64="$(plain_of "$console" | grep -oE 'trust_root=[A-Za-z0-9+/=]+' | head -1 | cut -d= -f2- || true)" + [[ -n "$b64" ]] && break + sleep 1 + done + [[ -n "$b64" ]] || fail "the enclave never reported a trust root; is S3FS_COSIGN_ATTESTATIONS set in the image?" + + TRUST_ROOT="$RUNDIR/trust-root.der" + printf '%s' "$b64" | base64 -d > "$TRUST_ROOT" \ + || { echo "the reported trust root is not base64" >&2; exit 1; } + openssl x509 -inform der -in "$TRUST_ROOT" -noout >/dev/null 2>&1 \ + || { echo "the reported trust root is not a certificate" >&2; exit 1; } + + # No `--allow-untrusted-root`, deliberately. With it, `verify` accepts + # whatever root a document arrived with and reports it as self-signed, so a + # "pinned" root pins nothing. Without it the presented root must equal this + # one — the same comparison a client makes against AWS's — and passing + # --pcr0 and --pcr16 becomes mandatory. + PASSKEY_ARGS=(--trust-root "$TRUST_ROOT" --pcr0 "$EXPECTED_PCR0" --pcr16 "$EXPECTED_PCR16" + --rp-id "${WEBAUTHN_RP_ID:-enclave.test}") +} + +# Nothing answers HTTPS until an ACME order completes — directory, account, +# order, challenge, finalize — so this loop is the issuance path finishing, not +# just a process starting. On a later boot the sealed cache has the certificate +# already, and it answers at once. +wait_for_https() { + local answered="" + # Bounded by TIMEOUT rather than a literal: a full ACME order on a slow + # machine is the longest wait here, and a fixed 90 was a CI-only failure + # waiting to be discovered on a busy runner. + for _ in $(seq "$TIMEOUT"); do + if curl -sk --max-time 2 "https://127.0.0.1:$HTTPS_PORT/" >/dev/null 2>&1; then + answered=yes + break + fi + sleep 1 + done + [[ -n "$answered" ]] +} + +# And it is genuinely the CA's, not something the enclave minted for itself. +# Pebble publishes the issuing chain on its management port, so this verifies +# the served chain against that root the way any PKI client would — which is the +# part a self-signed image could never test. +enclave_check_certificate() { + if [[ -n "$TLS_DOMAIN" ]]; then + enclave_check_public_certificate + return + fi + curl -sk --max-time 5 "https://127.0.0.1:15000/roots/0" > "$RUNDIR/pebble-root.pem" \ + || { echo "could not fetch Pebble's root" >&2; exit 1; } + curl -sk --max-time 5 "https://127.0.0.1:15000/intermediates/0" > "$RUNDIR/pebble-int.pem" \ + || { echo "could not fetch Pebble's intermediate" >&2; exit 1; } + if [[ -n "$KEEP_STORE" ]]; then + # Pebble mints a new CA every start, but a kept store holds the + # certificate an earlier Pebble issued and serves it until renewal. So + # every root and intermediate seen is kept, and clients are handed all + # of them: the served chain verifies whichever one issued it. + keep_pem "$RUNDIR/pebble-root.pem" "$STORE_DIR/pebble-roots.pem" + keep_pem "$RUNDIR/pebble-int.pem" "$STORE_DIR/pebble-ints.pem" + cp "$STORE_DIR/pebble-roots.pem" "$RUNDIR/pebble-root.pem" + cp "$STORE_DIR/pebble-ints.pem" "$RUNDIR/pebble-int.pem" + fi + echo | openssl s_client -connect "127.0.0.1:$HTTPS_PORT" -servername enclave.test -showcerts \ + 2>/dev/null | sed -n '/BEGIN CERT/,/END CERT/p' > "$RUNDIR/served-chain.pem" + [[ -s "$RUNDIR/served-chain.pem" ]] || { echo "the enclave presented no certificate" >&2; exit 1; } + + openssl verify -CAfile "$RUNDIR/pebble-root.pem" -untrusted "$RUNDIR/pebble-int.pem" \ + "$RUNDIR/served-chain.pem" \ + || { echo "the served certificate does not chain to the CA that issued it" >&2; exit 1; } + + # The name, from the SAN rather than the subject: an ACME certificate + # carries no CN at all — identity lives in subjectAltName — so printing the + # subject would print an empty string and look like a bug. + local names issuer + names="$(openssl x509 -in "$RUNDIR/served-chain.pem" -noout -ext subjectAltName \ + | tail -n +2 | tr -d ' ')" + issuer="$(openssl x509 -in "$RUNDIR/served-chain.pem" -noout -issuer)" + echo "served for $names" + echo "issued by $issuer" + [[ "$names" == *enclave.test* ]] \ + || { echo "the certificate is not for enclave.test: $names" >&2; exit 1; } + [[ "$issuer" == *Pebble* ]] \ + || { echo "the certificate was not issued by Pebble: $issuer" >&2; exit 1; } +} + +# A public name, from a public CA. Checked as any client on the internet checks it — the system's +# roots and the name — except against staging, whose roots nothing trusts by design; there the +# issuer is what says the order went where it should. +enclave_check_public_certificate() { + local out names issuer + out="$(echo | openssl s_client -connect "127.0.0.1:$HTTPS_PORT" -servername "$TLS_DOMAIN" \ + -verify_hostname "$TLS_DOMAIN" -showcerts 2>&1 || true)" + sed -n '/BEGIN CERT/,/END CERT/p' <<<"$out" > "$RUNDIR/served-chain.pem" + [[ -s "$RUNDIR/served-chain.pem" ]] || { echo "the enclave presented no certificate" >&2; exit 1; } + names="$(openssl x509 -in "$RUNDIR/served-chain.pem" -noout -ext subjectAltName | tail -n +2 | tr -d ' ')" + issuer="$(openssl x509 -in "$RUNDIR/served-chain.pem" -noout -issuer)" + echo "served for $names" + echo "issued by $issuer" + [[ "$names" == *"DNS:$TLS_DOMAIN"* ]] \ + || { echo "the certificate is not for $TLS_DOMAIN: $names" >&2; exit 1; } + if [[ -n "$ACME_STAGING" ]]; then + [[ "$issuer" == *STAGING* ]] \ + || { echo "expected a Let's Encrypt staging certificate: $issuer" >&2; exit 1; } + else + grep -q "Verify return code: 0 (ok)" <<<"$out" \ + || { echo "the served certificate does not verify against the system roots:" >&2 + grep "Verify return code" <<<"$out" >&2; exit 1; } + fi +} + +# Append the certificate in $1 to the bundle $2 unless it is already there. +keep_pem() { + touch "$2" + grep -qF "$(sed -n 2p "$1")" "$2" || cat "$1" >> "$2" +} + +# Everything above, in the one order that works. The ordering constraints are +# recorded at each step; this is the only place they are all satisfied at once. +enclave_bring_up() { + enclave_preflight + trap cleanup EXIT + + enclave_build_image + enclave_build_guest + enclave_build_clients + + enclave_start_store + enclave_start_parent + # A public name is validated by the public CA, over the internet, on this host's :443. + [[ -n "$TLS_DOMAIN" ]] || enclave_start_ca + enclave_expose_https + + say "booting the enclave" + boot_enclave "$PREFIX-qemu" "$CONSOLE" + + say "waiting for the enclave to come up" + wait_for_serving + plain | grep -E "enclave networking is up|guest measured into PCR16|mounted |serving +addr=" | tail -4 + + # Measured before anything asked for a key, and measured as the build said + # it would be. + grep -q "pcr16=$EXPECTED_PCR16" <(plain) \ + || fail "the enclave did not report measuring the expected guest into PCR16" + + enclave_trust_root + + local ca=Pebble + [[ -z "$TLS_DOMAIN" ]] || ca="Let's Encrypt" + say "waiting for $ca to issue the serving certificate" + wait_for_https || { + echo "no certificate was ever issued; the ACME path did not complete" >&2 + echo "--- pebble ---" >&2; docker logs "$PREFIX-pebble" 2>&1 | tail -30 >&2 + echo "--- console ---" >&2; plain | grep -i acme | tail -30 >&2 + exit 1 + } + echo "the enclave is serving a certificate it obtained over ACME" + enclave_check_certificate +} + +# Everything a run builds, in one directory, for a machine that should not build. +# +# The image is a file and its measurement is beside it; gvproxy is static; the verifier, the passkey +# client and vhost-device-vsock link only against the C library. So a host with Docker, KVM and +# vsock runs the whole stack from this with `--prebuilt` — no Nix store, no toolchain, no checkout +# that has to match — which is what makes a small VM enough. The image configuration is recorded +# beside it, because it is fixed by the image: a run cannot change it, only read it. +enclave_pack() { + enclave_preflight + enclave_build_image + enclave_build_clients + nix_build gvproxy "$RUNDIR/gvproxy" >/dev/null + GV_DIR="$(resolve_out_link "$RUNDIR/gvproxy")" + + say "packing into $PACK_DIR" + rm -rf "$PACK_DIR"; mkdir -p "$PACK_DIR/eif" "$PACK_DIR/bin" + cp -L "$EIF_DIR/s3fs-qemu.eif" "$EIF_DIR/pcr.json" "$PACK_DIR/eif/" + cp -L "$GV_DIR/bin/gvproxy" "$ATTEST" "$PASSKEY" "$VSOCK_BIN" "$PACK_DIR/bin/" + chmod -R u+w "$PACK_DIR" + { + printf 'WEBAUTHN_RP_ID=%q\n' "${WEBAUTHN_RP_ID:-}" + printf 'WEBAUTHN_ALLOWED_ORIGINS=%q\n' "${WEBAUTHN_ALLOWED_ORIGINS:-}" + printf 'GUEST_EGRESS_ORIGINS=%q\n' "${GUEST_EGRESS_ORIGINS:-}" + printf 'TLS_DOMAIN=%q\n' "$TLS_DOMAIN" + printf 'ACME_STAGING=%q\n' "$ACME_STAGING" + printf 'FCM_PROJECT=%q\n' "$FCM_PROJECT" + } > "$PACK_DIR/image.env" + echo "PCR0 $EXPECTED_PCR0" + du -sh "$PACK_DIR" | cut -f1 | sed 's/^/size /' +} diff --git a/deploy/qemu-nitro/pebble/README.md b/deploy/qemu-nitro/pebble/README.md new file mode 100644 index 0000000..eff6920 --- /dev/null +++ b/deploy/qemu-nitro/pebble/README.md @@ -0,0 +1,59 @@ +# Pebble's own TLS material, for the QEMU e2e + +Pebble is the ACME test CA the end-to-end runs against. Two different +certificates are involved and they are easy to confuse: + +- **The certificates Pebble *issues*** — the enclave's serving certificate. + Pebble generates a fresh root and intermediate at every startup and publishes + them on its management port, so nothing about those is checked in. +- **The certificate Pebble *serves its own API under*** — that is what is here. + The enclave has to validate it before it will send an ACME order, and it is + reached at `https://192.168.127.254:14000/dir`, gvproxy's host address. + +Pebble's built-in certificate is `CN=localhost` with no SAN for that address, +so the harness supplies its own. + +| | | +|---|---| +| `ca.pem` | The root. Baked into `eif-qemu` at `/pebble-ca.pem` and named by `S3FS_ACME_CA`, so it is covered by PCR0 like any other configuration. | +| `cert.pem`, `key.pem` | Pebble's API certificate, SANs `192.168.127.254`, `127.0.0.1`, `localhost`, `pebble`. Mounted into the container. | + +The CA's private key is **deliberately not here**: it was destroyed after +signing, so this root can never issue anything else. Both certificates are +valid for 100 years, which keeps `eif-qemu` reproducible — regenerating them on +each build would change PCR0 every time and make the measurement meaningless. + +None of this is a secret and none of it is trusted by anything but the QEMU +harness. The production image ships no such file, and a production binary has +no `--acme-ca` flag to read one with. + +To regenerate (only if a SAN must change): + +```console +$ cd deploy/qemu-nitro/pebble +$ openssl req -x509 -newkey ec -pkeyopt ec_paramgen_curve:P-256 -nodes \ + -keyout ca-key.pem -out ca.pem -days 36500 \ + -subj "/CN=s3fs e2e Pebble test CA" \ + -addext "basicConstraints=critical,CA:TRUE,pathlen:0" \ + -addext "keyUsage=critical,keyCertSign,cRLSign" +$ openssl req -newkey ec -pkeyopt ec_paramgen_curve:P-256 -nodes \ + -keyout key.pem -out csr.pem -subj "/CN=pebble.e2e" +$ cat > san.cnf <<'CNF' +[ext] +basicConstraints = critical, CA:FALSE +keyUsage = critical, digitalSignature, keyEncipherment +extendedKeyUsage = serverAuth +subjectAltName = @alt +[alt] +IP.1 = 192.168.127.254 +IP.2 = 127.0.0.1 +DNS.1 = localhost +DNS.2 = pebble +CNF +$ openssl x509 -req -in csr.pem -CA ca.pem -CAkey ca-key.pem -CAcreateserial \ + -days 36500 -extfile san.cnf -extensions ext -out cert.pem +$ rm -f ca-key.pem ca.srl csr.pem san.cnf # the CA key does not survive +``` + +PCR0 changes when `ca.pem` does, so `run-e2e.sh` will report the new +measurement and leg 4 keeps checking it against what `nix build` printed. diff --git a/deploy/qemu-nitro/pebble/ca.pem b/deploy/qemu-nitro/pebble/ca.pem new file mode 100644 index 0000000..300bed8 --- /dev/null +++ b/deploy/qemu-nitro/pebble/ca.pem @@ -0,0 +1,12 @@ +-----BEGIN CERTIFICATE----- +MIIBrjCCAVSgAwIBAgIUQD12wUykI/U6/VrFj8i8oB2SfSkwCgYIKoZIzj0EAwIw +IjEgMB4GA1UEAwwXczNmcyBlMmUgUGViYmxlIHRlc3QgQ0EwIBcNMjYwOTA1MTcw +MzU2WhgPMjEyNjA4MTIxNzAzNTZaMCIxIDAeBgNVBAMMF3MzZnMgZTJlIFBlYmJs +ZSB0ZXN0IENBMFkwEwYHKoZIzj0CAQYIKoZIzj0DAQcDQgAExCTDvZXgOZNbBXqH +GG2BWrLDUmPT+ulf5fEVVoUpAOMZ9c1atvb3eaZWCymCfDwGXu8oVJzdufkvg7Md +vsVXGqNmMGQwHQYDVR0OBBYEFLL7+ljJpvqI5YJsguK2EVFmnDpkMB8GA1UdIwQY +MBaAFLL7+ljJpvqI5YJsguK2EVFmnDpkMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYD +VR0PAQH/BAQDAgEGMAoGCCqGSM49BAMCA0gAMEUCIEnwRJzFrZfCfjfes2/NZvME +Y6OnwgGv/hOJMZj1Wg4KAiEA3LMfpt7o53eehVzmTc8Ql/5dQl8odlrhmA1tzvTB +0rc= +-----END CERTIFICATE----- diff --git a/deploy/qemu-nitro/pebble/cert.pem b/deploy/qemu-nitro/pebble/cert.pem new file mode 100644 index 0000000..308eabb --- /dev/null +++ b/deploy/qemu-nitro/pebble/cert.pem @@ -0,0 +1,13 @@ +-----BEGIN CERTIFICATE----- +MIIB3TCCAYKgAwIBAgIUXe6OgSa9VDTibtka7c2sJddaQf8wCgYIKoZIzj0EAwIw +IjEgMB4GA1UEAwwXczNmcyBlMmUgUGViYmxlIHRlc3QgQ0EwIBcNMjYwOTA1MTcw +MzU2WhgPMjEyNjA4MTIxNzAzNTZaMBUxEzARBgNVBAMMCnBlYmJsZS5lMmUwWTAT +BgcqhkjOPQIBBggqhkjOPQMBBwNCAAR+xvCwK9GQppz7A8C7RLvDLsSHotC/sYPd +UepEN4XIPPkFn8rBGtjY71wOJnzT27jzqzBdlTevayCrRY67jU7uo4GgMIGdMAwG +A1UdEwEB/wQCMAAwDgYDVR0PAQH/BAQDAgWgMBMGA1UdJQQMMAoGCCsGAQUFBwMB +MCgGA1UdEQQhMB+HBMCof/6HBH8AAAGCCWxvY2FsaG9zdIIGcGViYmxlMB0GA1Ud +DgQWBBSINS8mA66g7mJguh9a4PuU3qnkOjAfBgNVHSMEGDAWgBSy+/pYyab6iOWC +bILithFRZpw6ZDAKBggqhkjOPQQDAgNJADBGAiEApDJylkqE2vdNfnpDg8LKV0UI +rtxZpJuftF7HT6vqmLcCIQDrYGTlGernCX3fYu/0GCYV+l3WkPElJP+p1ClNFmyy +rQ== +-----END CERTIFICATE----- diff --git a/deploy/qemu-nitro/pebble/key.pem b/deploy/qemu-nitro/pebble/key.pem new file mode 100644 index 0000000..74c1883 --- /dev/null +++ b/deploy/qemu-nitro/pebble/key.pem @@ -0,0 +1,5 @@ +-----BEGIN PRIVATE KEY----- +MIGHAgEAMBMGByqGSM49AgEGCCqGSM49AwEHBG0wawIBAQQgy1r1ChPj8Mjovu4p +d15QQu85KDldIOIOI8MuQr7+eRKhRANCAAR+xvCwK9GQppz7A8C7RLvDLsSHotC/ +sYPdUepEN4XIPPkFn8rBGtjY71wOJnzT27jzqzBdlTevayCrRY67jU7u +-----END PRIVATE KEY----- diff --git a/deploy/qemu-nitro/run-e2e.sh b/deploy/qemu-nitro/run-e2e.sh new file mode 100755 index 0000000..4688e16 --- /dev/null +++ b/deploy/qemu-nitro/run-e2e.sh @@ -0,0 +1,544 @@ +#!/usr/bin/env bash +# The whole stack, in an emulated enclave. +# +# Everything below has been verified in pieces — the block store against MinIO, +# TLS and the attestation binding in unit tests, NSM entropy under QEMU. None +# of it had ever run together, because until gvproxy the emulated enclave had +# no way to reach anything. +# +# host QEMU enclave +# ──── ──────────── +# MinIO :9000 ◀── gvproxy ──192.168.127.254── s3fs mount, and the guest +# gvproxy --listen vsock://:1024 ────────────▶ gvforwarder → tap0 .2 +# expose :8443 → 192.168.127.2:443 ──▶ rustls :443 +# vhost-device-vsock --forward-cid 1 +# heartbeat.py :9000 ───────────────────────▶ init's boot heartbeat +# nitro-attest ──────────────────────────────▶ x-enclave-attestation +# +# What it proves, in order of how much it cost to get here: +# +# 1. PCR0 in the signed attestation document equals the PCR0 `nix build` +# printed. The measurement a client would pin is the measurement the +# reproducible build claimed. +# 2. user_data binds the certificate from this connection's own handshake, +# so the TLS session terminates in the attested enclave. +# 3. The guest's counter advances, so writes crossed gvproxy to MinIO and +# came back — the filesystem really is mounted over the emulated vsock. +# 4. The guest came from the store, not the image. The enclave measured the +# object it fetched into PCR16 and locked it; the attested PCR16 is the one +# the release build computed; and a substituted object boots an enclave +# that a client pinning the approved guest refuses. +# +# What this harness CANNOT prove, stated up front because it is easy to assume +# otherwise. Every client here runs its full verification path — COSE ES384, +# the certificate chain, the validity windows, the pinned root, both PCRs — and +# that is real: it is the same code, with the same flags, that will run against +# hardware. What it is not is proof that a *Nitro enclave* produced anything. +# +# QEMU's NSM does not sign at all. Its source says so — "we don't actually sign +# the data, so we use -1 as the 'alg' value" — and -1 is not a COSE algorithm +# identifier. The emulator image therefore mints a chain at boot and re-signs +# the documents the device produced, contents untouched. The key lives inside an +# image whoever boots it controls, so a verified document here means "this image +# said so", where on hardware it means "a Nitro enclave with this measurement +# said so". That gap needs hardware and nothing here can close it. +# +# So the contents are what this harness is really checking, and they are the +# part that is our code: the nonce the client asked for, the PCR0 of the image +# running, the PCR16 of the guest measured, and the hash of the certificate +# being served. KMS refusing a substituted guest is the other thing that needs +# real hardware — the emulator image cannot use KMS at all, because KMS will not +# accept a document it did not see a Nitro root behind. +set -euo pipefail + +# The bring-up is shared with `dev-enclave.sh`, which stands up exactly this +# stack and then leaves it running instead of asserting against it. Everything +# up to "the enclave is serving" lives in lib.sh for that reason. +PREFIX=e2e +# A second guest, altered, for leg 8. Nothing else here needs one, so lib.sh +# does not stage it unless asked. +WITH_SUBSTITUTE=1 +# shellcheck source=lib.sh +source "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/lib.sh" + +enclave_bring_up + +# --------------------------------------------------------------------------- +# The assertions. +# --------------------------------------------------------------------------- +say "1/8 the guest is unreachable without a passkey assertion" +# The rule, at the front because everything after it depends on it holding: +# nothing reaches the guest without a fresh assertion bound to that request. +# +# The nonce is sent so the *gate* is what refuses. Every request needs one, +# and it is checked before routing — so without it these would be 400s, and +# this leg would pass while proving nothing about the gate. +nonce() { openssl rand 20 | basenc --base64url | tr -d '='; } +for path in / /counter /memory; do + code="$(curl -sk -o /dev/null -w '%{http_code}' --max-time 20 \ + -H "x-enclave-nonce: $(nonce)" \ + "https://127.0.0.1:$HTTPS_PORT$path")" + [[ "$code" == "401" ]] \ + || fail "$path answered $code without an assertion; the gate is not wired up" +done +echo "unauthenticated requests refused: 401" + +# And a request with no nonce at all never reaches the gate either. +code="$(curl -sk -o /dev/null -w '%{http_code}' --max-time 20 \ + "https://127.0.0.1:$HTTPS_PORT/counter")" +[[ "$code" == "400" ]] \ + || fail "a request with no nonce answered $code; it should be refused before routing" +echo "un-nonced requests refused: 400" + +say "2/8 a passkey enrols and its signed requests reach the guest" +rm -f "$RUNDIR/alice.json" "$RUNDIR/bob.json" +signed enrol >/dev/null || fail "registration failed" + +first="$(signed get --path /counter)" || fail "no answer from the guest" +second="$(signed get --path /counter)" || fail "no answer from the guest" +echo "counter: $first then $second" +[[ "${second//[^0-9]/}" -eq $(( ${first//[^0-9]/} + 1 )) ]] \ + || fail "the counter did not advance ($first → $second); writes are not reaching MinIO" + +say "2b/8 signed requests drive a real amount of filesystem work" +# Two round trips to /counter prove the mount answers. They do not prove much +# about the filesystem underneath it, which is the part with a Merkle block +# store, encryption and a root record behind it. +# +# So: a spread of files written, read back, overwritten and read again, each one +# a separate signed interaction — which is also the only way this harness can +# exercise the store, since every request has to pass the gate. Enough to force +# real block allocation and rewriting; not so much that the ceremony cost +# dominates the run. +FILES=8 +for i in $(seq 1 "$FILES"); do + signed post --path "/files/work-$i.txt" --body "contents $i" >/dev/null \ + || fail "writing /files/work-$i.txt failed" +done +for i in $(seq 1 "$FILES"); do + got="$(signed get --path "/files/work-$i.txt")" || fail "reading /files/work-$i.txt failed" + [[ "$got" == "contents $i" ]] \ + || fail "/files/work-$i.txt read back as \"$got\"" +done +# Overwriting in place is the case that rewrites blocks rather than appending +# new ones, and the one a copy-on-write store can get wrong while every fresh +# write still looks correct. +for i in $(seq 1 "$FILES"); do + signed post --path "/files/work-$i.txt" --body "rewritten $i" >/dev/null \ + || fail "overwriting /files/work-$i.txt failed" +done +for i in $(seq 1 "$FILES"); do + got="$(signed get --path "/files/work-$i.txt")" || fail "re-reading /files/work-$i.txt failed" + [[ "$got" == "rewritten $i" ]] \ + || fail "/files/work-$i.txt kept a stale value \"$got\" after being overwritten" +done +echo "$FILES files written, read, overwritten and re-read through the gate" + +say "3/8 the attestation binds this connection's certificate, runtime and guest" +# There is no attestation endpoint: the document rides on the `/auth/` exchange, +# in `x-enclave-attestation`. That is where a client identifies the enclave +# before approving anything, and it needs no credential — so this is the check a +# client makes on the connection it then goes on to use. +"$ATTEST" \ + --url "https://127.0.0.1:$HTTPS_PORT/auth/" \ + --trust-root "$TRUST_ROOT" \ + --pcr0 "$EXPECTED_PCR0" \ + --guest "$RUNDIR/guests/guest.wasm" \ + | tee "$RUNDIR/attest.log" \ + || fail "attestation verification failed" + +grep -q "binding the attested certificate" "$RUNDIR/attest.log" \ + || fail "the document did not bind the certificate this connection was served" + +say "3b/8 a signed interaction verifies the enclave before it trusts it" +# The proof comes from the `/auth/` exchange, not from the guest's response: +# that exchange is where a client identifies the enclave, before it hands over +# an assertion and before the interaction runs. Guest responses deliberately +# carry no document — the client has pinned the certificate by then, and TLS +# proves the peer still holds its key. +# +# `--dump-proof` keeps the three things a verifier cannot recover afterwards: +# the document, the certificate that connection was served, and the nonce the +# request sent. +rm -rf "$RUNDIR/proof" +signed --dump-proof "$RUNDIR/proof" get --path /counter >/dev/null \ + || fail "the signed interaction failed" +[[ -s "$RUNDIR/proof/document.b64" ]] \ + || fail "the challenge exchange carried no attestation document" + +"$ATTEST" \ + --document "$RUNDIR/proof/document.b64" \ + --peer-certificate "$RUNDIR/proof/certificate.der" \ + --nonce "$(cat "$RUNDIR/proof/nonce.hex")" \ + --trust-root "$TRUST_ROOT" \ + --pcr0 "$EXPECTED_PCR0" \ + --guest "$RUNDIR/guests/guest.wasm" \ + | tee "$RUNDIR/attest-guest.log" \ + || fail "the challenge exchange's document did not verify" + +grep -q "binding the attested certificate" "$RUNDIR/attest-guest.log" \ + || fail "the document did not bind that connection's certificate" + +say "4/8 the attested PCR0 and PCR16 are the ones the builds produced" +# `nitro-attest --pcr0 --guest` already enforced both, so reaching here means +# they held. Printing them is what makes the claim checkable by eye rather than +# taken on trust from an exit code. +echo "PCR0 build: $EXPECTED_PCR0" +echo " attested: $(grep -oE '^PCR0 +[0-9a-f]+' "$RUNDIR/attest.log" | awk '{print $2}')" +echo "PCR16 release: $EXPECTED_PCR16" +echo " attested: $(grep -oE '^PCR16 +[0-9a-f]+' "$RUNDIR/attest.log" | awk '{print $2}')" + +# --------------------------------------------------------------------------- +# 5/8 — one instance per tenant, and one approval per interaction. +# --------------------------------------------------------------------------- +say "5/8 a tenant keeps its instance, and no two tenants share one" + +# `/memory` counts in the guest's linear memory and writes nowhere. What it +# answers is the whole per-tenant model in one number. +# +# The image runs with warm instances, so the *same* tenant asking twice must +# see the count rise — that is the instance being kept. A *different* tenant +# must see 1, because the boundary between two clients is a `Store` and not +# anything the guest does. An earlier version of this leg asserted the +# opposite, having been written before warm instances existed; it contradicted +# the image it was testing. +m1="$(signed get --path /memory)" +m2="$(signed get --path /memory)" +echo "alice memory: $m1 then $m2" +[[ "${m2//[^0-9]/}" -eq $(( ${m1//[^0-9]/} + 1 )) ]] \ + || fail "a tenant's instance was not kept between its requests ($m1, $m2)" + +signed2 enrol >/dev/null || fail "the second registration failed" +b1="$(signed2 get --path /memory)" +echo "bob memory: $b1" +[[ "${b1//[^0-9]/}" -eq 1 ]] \ + || fail "a second tenant landed in the first tenant's instance ($b1)" + +# And their storage is separate too: the same path, different contents. +signed post --path /files/who.txt --body "alice" >/dev/null || fail "alice could not write" +signed2 post --path /files/who.txt --body "bob" >/dev/null || fail "bob could not write" +a_sees="$(signed get --path /files/who.txt)" +b_sees="$(signed2 get --path /files/who.txt)" +echo "alice reads: $a_sees / bob reads: $b_sees" +[[ "$a_sees" == "alice" && "$b_sees" == "bob" ]] \ + || fail "one tenant read another's file (alice=$a_sees bob=$b_sees)" + +say "5b/8 an approval for one route does not authorize another" +# The property the interaction token exists for. A token names the interaction +# it was issued for — method, path and query — so spent on any other it is +# refused, and what it carried never reaches the filesystem. +# +# Captured whole and split here rather than piped through `head`: `head` exits +# after one line, and a client still writing the rest dies of SIGPIPE, which +# `pipefail` would report as this leg failing. +sub_out="$(signed substitute --approved /files/approved.txt \ + --sent /files/substituted.txt --body "substituted")" +sub="${sub_out%%$'\n'*}" +[[ "$sub" == "401" ]] || fail "a token was spent on a route it was not issued for (status $sub)" +if signed get --path /files/substituted.txt >/dev/null 2>&1; then + fail "the substituted request reached the filesystem" +fi +echo "a token moved to another route was refused: 401, and nothing was written" + +# --------------------------------------------------------------------------- +# 5c/8 — guest output, in a real enclave. +# --------------------------------------------------------------------------- +# The guest's stdout and stderr are no longer inherited: the runtime frames them +# into lines and emits them as its own structured events. Unit tests prove the +# framing and integration tests prove the wiring; only here is it running inside +# the enclave, on the console the parent actually reads. +# +# What matters is that guest text arrives *marked as guest text*. It is chosen +# by the guest, so it must never be mistakable for something the runtime said. +say "5c/8 guest output reaches the console tagged as untrusted" +signed get --path /log >/dev/null || fail "the guest refused to log" + +# Wait for the guest's *last* line, not its first. +# +# `docker logs` fills this console asynchronously, so "some guest output has +# arrived" says nothing about the rest of it — and every assertion below is +# about a line the guest wrote later. Waiting on the first line and then +# grepping for the third is a race that passes on a quiet machine and fails on +# a busy one, which is how it was found. +# +# The unterminated tail is last: it is emitted when the stream object drops, +# after the response, and after stderr was flushed during the request. Once it +# is here, everything else already is. +tail_seen="" +for _ in $(seq "$TIMEOUT"); do + grep -q 'guest_message="no trailing newline"' <(plain) && { tail_seen=1; break; } + sleep 1 +done +[[ -n "$tail_seen" ]] || fail "the guest's unterminated last line never arrived" + +# Two guest writes joined into one line, and CRLF normalised. +grep -q 'guest_message="first line"' <(plain) \ + || fail "two guest writes were not joined into one line" +grep -q 'guest_message="windows"' <(plain) \ + || fail "CRLF was not normalised" +# A blank line the guest wrote is still a record. +grep -q 'guest_message=""' <(plain) \ + || fail "the guest's empty line was dropped" + +# The distinction the design rests on, kept all the way to the console. +grep -q 'guest_stream="stdout".*guest_message="first line"' <(plain) \ + || fail "guest stdout was not tagged as stdout" +grep -q 'guest_stream="stderr".*guest_message="on stderr"' <(plain) \ + || fail "guest stderr was not tagged as stderr" + +# Untrusted text must be *visibly* untrusted, not merely filterable. Runtime +# events name a module in this runtime; guest output names `guest`, and nothing +# a guest writes can change which target its line carries. +if plain | grep 'guest output' | grep -qv ' guest: guest output'; then + echo "--- offending lines ---" >&2 + plain | grep 'guest output' | grep -v ' guest: guest output' | head -5 >&2 + fail "a guest line reached the console without the guest target" +fi +# Guest text is one quoted, escaped field value. It was not always: naming the +# field `message` collided with the event's own message and printed guest bytes +# bare in the structured part of the line, where `truncated=true` from a guest +# rendered as a field nobody set. This is that fix, held in place. +if plain | grep 'guest output' | grep -qvE 'guest_message="'; then + fail "guest output reached the console outside a quoted field" +fi + +grep -q 'enclave_runtime::' <(plain) \ + || fail "no runtime event carried a module target to be distinguished from" +if plain | grep 'enclave_runtime::' | grep -q ' guest: '; then + fail "a runtime event carried the guest target" +fi + +echo "guest output arrived framed, tagged by stream, and marked as guest" + + +# --------------------------------------------------------------------------- +# 6/8 — work that outlives the interaction that asked for it. +# --------------------------------------------------------------------------- +# Everything above is request and response: a client signs, the guest answers, +# and the approval is spent by the time the connection closes. Background work +# is the one place that shape does not hold. The assertion authorises an +# enqueue, and what it authorised runs later on the enclave's own schedule — +# with nobody signing anything at that moment, because there is no one there to +# sign. That is the whole of standing authority, and it is the part a +# request/response test cannot reach. +# +# Worth proving in an enclave rather than only in-process: the scheduler runs +# against the mounted filesystem and the per-tenant lock, and neither of those +# is what a unit test exercises. +say "6/8 work approved once runs later, without a second assertion" + +# Enrolled first, or there is nobody to wake when the task finishes. This is an +# ordinary signed interaction: enrolling is interactive-only, so it could not +# have been done by the background work itself. +FCM_TOKEN="e2e-device:APA91bEnclaveRuntimeHarnessToken0123456789" +signed post --path /devices --body "$FCM_TOKEN" >/dev/null \ + || fail "enrolling a device was refused" +[[ "$(signed get --path /devices)" == "1" ]] \ + || fail "the device did not enrol" + +signed post --path /tasks/e2e-job --body "scheduled work" >/dev/null \ + || fail "the scheduled task was refused" + +# The work belongs to the passkey that asked for it. Bob holds a perfectly good +# credential and is still told there is no such task, because a task id is +# scoped to its tenant rather than being a name everyone shares. +if signed2 get --path /tasks/e2e-job >/dev/null 2>&1; then + fail "a second tenant could see another tenant's task" +fi + +# Polled, not slept on. The first occurrence is due immediately, but +# "immediately" still means a worker picking it up, instantiating the guest, and +# writing the result through the filesystem to MinIO. +completed="" +for _ in $(seq "$TIMEOUT"); do + record="$(signed get --path /tasks/e2e-job)" || fail "the task record became unreadable" + case "$(jq -r .status <<<"$record")" in + completed) completed=1; break ;; + failed) fail "the scheduled task failed: $record" ;; + esac + sleep 1 +done +[[ -n "$completed" ]] || fail "the scheduled task never ran within ${TIMEOUT}s" + +# And it ran the work actually asked for, rather than merely reaching a terminal +# state. The result is the payload the guest echoed back, carried as bytes. +result="$(jq -r '.result | implode' <<<"$record")" +[[ "$result" == "scheduled work" ]] \ + || fail "the task completed but produced \"$result\"" +echo "scheduled work ran for its owner alone, with no second assertion signed" + +say "6b/8 the finished task woke its owner, and told Google nothing" +# The wake was raised inside `run-task`, by background work, with nobody signing +# anything at that moment — and it left the enclave as a *data-only* message. +# +# That absence is the property: an FCM payload crosses the parent instance and +# then Google, so a title or a body would disclose to both exactly what this +# enclave exists to keep from them. The app wakes and fetches the detail over +# its own attested connection. +recorded="" +for _ in $(seq "$TIMEOUT"); do + [[ -s "$FCM_RECORD" ]] && { recorded=1; break; } + sleep 1 +done +[[ -n "$recorded" ]] || fail "the finished task never woke anybody" + +wake="$(head -1 "$FCM_RECORD")" +echo "$wake" | jq -e '.message.notification == null' >/dev/null \ + || fail "the wake carried a notification block: $wake" +[[ "$(echo "$wake" | jq -r '.message.data.category')" == "task-done" ]] \ + || fail "unexpected category: $wake" +[[ "$(echo "$wake" | jq -r '.message.data.ref')" == "e2e-job" ]] \ + || fail "the wake did not name the task: $wake" +[[ "$(echo "$wake" | jq -r '.message.token')" == "$FCM_TOKEN" ]] \ + || fail "the wake went to a device nobody enrolled: $wake" +[[ "$(echo "$wake" | jq -r '.message.apns.payload.aps."content-available"')" == "1" ]] \ + || fail "the wake would not have woken an iOS app: $wake" + +# And nothing a person would read went with it. The only strings in `data` are +# the two labels the guest chose and a schema version. +keys="$(echo "$wake" | jq -r '.message.data | keys | join(",")')" +[[ "$keys" == "category,ref,v" ]] || fail "the wake carried more than its labels: $keys" +echo "a data-only wake reached the enrolled device: category=task-done ref=e2e-job" + +# --------------------------------------------------------------------------- +# 7/8 — the boot machine, across a restart. +# --------------------------------------------------------------------------- +# The first boot found an empty store and created a filesystem. That used to be +# what happened for *any* store that answered "nothing", including one whose +# contents had been hidden. The second boot has to recognise the state as its +# own and resume — which is only possible if the receipt the first boot wrote +# verifies against the state now present — and, running the same guest, find +# its pair record and write nothing. +say "7/8 a second boot resumes rather than starting over" + +# `docker logs -f` fills the console file asynchronously, so a single grep can +# run before the line it is looking for has been written — the assertions above +# reach the enclave over the network and do not wait for its console. Poll, +# with a bound, the way the readiness check above already does. +genesis="" +for _ in $(seq "$TIMEOUT"); do + grep -q 'mode=Genesis' <(plain) && { genesis=1; break; } + sleep 1 +done +[[ -n "$genesis" ]] || fail "the first boot should have been a genesis" +GENESIS_CONSOLE="$CONSOLE" + +docker rm -f "$PREFIX-qemu" >/dev/null 2>&1 || true +sleep 2 + +CONSOLE="$RUNDIR/console-resume.log" +boot_enclave "$PREFIX-qemu-resume" "$CONSOLE" +# A new boot mints a new signing chain, so the root the last one reported is +# now the wrong one. Refreshed here rather than at first use, so a client run +# against this enclave fails on what it is checking and not on a stale pin. +enclave_trust_root + +resumed="" +for _ in $(seq "$TIMEOUT"); do + grep -q "state origin established" <(plain) && { resumed=1; break; } + grep -qE "Kernel panic|failed to start the guest" <(plain) && break + sleep 1 +done +[[ -n "$resumed" ]] || fail "the second boot never established a state origin" + +plain | grep -E "state origin established" | tail -1 +grep -q "mode=Resume" <(plain) \ + || fail "the second boot did not resume — it should not have created anything" +grep -q "pcr16=$EXPECTED_PCR16" <(plain) \ + || fail "the second boot did not measure the same guest" + +# Same filesystem, same identity: the receipt names this state and no other. +first_root="$(plain_of "$GENESIS_CONSOLE" | grep -oE 'state_root=[0-9a-f]+' | head -1)" +second_root="$(plain | grep -oE 'state_root=[0-9a-f]+' | head -1)" +echo "genesis $first_root" +echo "resume $second_root" +[[ "$first_root" == "$second_root" ]] \ + || fail "the state_root changed across a restart" + +# --------------------------------------------------------------------------- +# 8/8 — a substituted guest. +# --------------------------------------------------------------------------- +# The parent controls the store the guest is fetched from, so it can replace the +# object. What it cannot do is make the replacement look like the approved +# guest: the enclave measures whatever arrives into PCR16 before anything asks +# for a key. That is shown here from both sides — the enclave records a runtime +# and guest this state has not held before, and a client pinning the approved +# guest refuses to talk to it. +# +# What this cannot show is KMS refusing it, which is the half that keeps the +# data out of reach. The emulator image uses the static key source, because KMS +# will not accept an unsigned document, so this enclave boots and serves. On +# hardware, under a policy pinning the approved PCR16, the same substitution +# gets no key and reads nothing. +say "8/8 a substituted guest is measured, recorded, and refused by a pinned client" +docker rm -f "$PREFIX-qemu-resume" >/dev/null 2>&1 || true +upload_guest substitute.wasm +sleep 2 + +CONSOLE="$RUNDIR/console-substitute.log" +boot_enclave "$PREFIX-qemu-substitute" "$CONSOLE" +wait_for_serving +enclave_trust_root + +grep -q "pcr16=$SUBSTITUTE_PCR16" <(plain) \ + || fail "the enclave did not measure the object it fetched into PCR16" +grep -q "mode=Upgrade" <(plain) \ + || fail "a guest this state had never held was not recorded as an upgrade" +wait_for_https || fail "the enclave running the substitute never answered HTTPS" + +if "$ATTEST" --url "https://127.0.0.1:$HTTPS_PORT/auth/" --trust-root "$TRUST_ROOT" \ + --pcr0 "$EXPECTED_PCR0" --guest "$RUNDIR/guests/guest.wasm" \ + > "$RUNDIR/attest-substitute.log" 2>&1; then + fail "a client pinning the approved guest accepted an enclave running another" +fi +grep -q "PCR16 mismatch" "$RUNDIR/attest-substitute.log" \ + || fail "the client refused, but not on PCR16: $(tail -1 "$RUNDIR/attest-substitute.log")" +echo "a client pinning the approved guest refused: PCR16 mismatch" + +# And it is not hiding what it runs: pinned to the substitute, the same client +# accepts. The enclave attests the guest it fetched, whichever that was. +"$ATTEST" --url "https://127.0.0.1:$HTTPS_PORT/auth/" --trust-root "$TRUST_ROOT" \ + --pcr0 "$EXPECTED_PCR0" --guest "$RUNDIR/guests/substitute.wasm" \ + > "$RUNDIR/attest-substitute-pinned.log" \ + || fail "the enclave does not attest the guest it actually fetched" +echo "it attests the substitute it is running: PCR16 $SUBSTITUTE_PCR16" + +docker rm -f "$PREFIX-qemu-substitute" >/dev/null 2>&1 || true + +cat </dev/null || { echo "nix is not on PATH; see deploy/nix/README.md" >&2; exit 1; } +nix build "$REPO#eif-selftest" --out-link "$WORK/eif-selftest" 2>&1 | tail -2 +EIF_DIR="$(readlink -f "$WORK/eif-selftest" 2>/dev/null || true)" +# nix-portable keeps its store outside /nix except inside its own namespace; on +# a normal Nix install the first branch always wins. +[[ -d "$EIF_DIR" ]] || EIF_DIR="$HOME/.nix-portable$(readlink "$WORK/eif-selftest")" +EIF="$EIF_DIR/selftest.eif" +[[ -f "$EIF" ]] || { echo "cannot resolve the built EIF" >&2; exit 1; } +[[ -e /dev/kvm ]] || { echo "no /dev/kvm — the nitro-enclave machine needs KVM" >&2; exit 1; } + +rm -rf "$RUNDIR"; mkdir -p "$RUNDIR" + +VSOCK_BIN="$WORK/tools/bin/vhost-device-vsock" +[[ -x "$VSOCK_BIN" ]] || { echo "missing $VSOCK_BIN (cargo install vhost-device-vsock)" >&2; exit 1; } + +pids=() +cleanup() { + for p in "${pids[@]:-}"; do kill "$p" 2>/dev/null || true; done + wait 2>/dev/null || true +} +trap cleanup EXIT + +[[ -e /dev/vsock ]] || { + cat >&2 <<'EOF' +no /dev/vsock — the host needs vsock_loopback loaded: + sudo modprobe vsock_loopback +The enclave init dials CID 3, which vhost-device-vsock's unix-socket backend +refuses to route ("dropping packet for unknown cid: 3"), so the heartbeat has +to be answered on a real host vsock. +EOF + exit 1 +} + +# Port 9000 is where init expects the parent. +python3 "$REPO/deploy/qemu-nitro/heartbeat.py" 9000 \ + > "$RUNDIR/heartbeat.log" 2>&1 & +pids+=($!) + +# forward-cid=1 turns guest-originated connections into host vsock connections +# on the loopback CID, which is the only arrangement that reaches a listener +# for CID 3. It is mutually exclusive with --uds-path. +RUST_LOG="${VSOCK_LOG:-info}" "$VSOCK_BIN" \ + --guest-cid 4 \ + --socket "$RUNDIR/vhost.socket" \ + --forward-cid 1 \ + > "$RUNDIR/vsock.log" 2>&1 & +pids+=($!) + +for _ in $(seq 50); do [[ -S "$RUNDIR/vhost.socket" ]] && break; sleep 0.1; done +[[ -S "$RUNDIR/vhost.socket" ]] || { echo "vhost-device-vsock never came up:" >&2; cat "$RUNDIR/vsock.log" >&2; exit 1; } + +echo "== booting $EIF ==" +# --network none is deliberate: an enclave has no NIC, and the harness should +# not quietly give the guest one. +set +e +timeout "$TIMEOUT" docker run --rm \ + --device /dev/kvm \ + --network none \ + -v "$EIF_DIR:/eif:ro" \ + -v "$RUNDIR:/run/vsock" \ + "$IMAGE" \ + qemu-system-x86_64 \ + -M nitro-enclave,vsock=chr0,id=selftest \ + -kernel /eif/selftest.eif \ + -chardev socket,id=chr0,path=/run/vsock/vhost.socket \ + -m 1G -smp 2 -nographic -no-reboot \ + 2>&1 | tee "$CONSOLE" +set -e + +echo +echo "== result ==" +if grep -q "NSM-SELFTEST-OK" "$CONSOLE"; then + echo "PASS: the guest drew random bytes from the emulated NSM" + grep -E "NSM-SELFTEST|nsm" "$CONSOLE" || true + exit 0 +fi + +echo "FAIL: no NSM-SELFTEST-OK on the console" >&2 +echo "--- heartbeat ---" >&2; cat "$RUNDIR/heartbeat.log" >&2 +echo "--- vsock ---" >&2; tail -20 "$RUNDIR/vsock.log" >&2 +exit 1 diff --git a/deploy/tofu/.terraform.lock.hcl b/deploy/tofu/.terraform.lock.hcl new file mode 100644 index 0000000..08b75e9 --- /dev/null +++ b/deploy/tofu/.terraform.lock.hcl @@ -0,0 +1,29 @@ +# This file is maintained automatically by "tofu init". +# Manual edits may be lost in future updates. + +provider "registry.opentofu.org/hashicorp/aws" { + version = "5.100.0" + constraints = "~> 5.0" + hashes = [ + "h1:7/GgVlN+KplSVCuc8qb4ct2R7gotYooPNRd0cnj9GxE=", + "h1:BrNG7eFOdRrRRbHdvrTjMJ8X8Oh/tiegURiKf7J2db8=", + "h1:C6eM6fGJVktK2M5vH3Yhv5NnqmegcBDY0EuDHhiXoVY=", + "h1:C7yD4Be2zhVdjnilsKPfucYAYMG5UCJYuUSoY6FCtGQ=", + "h1:H8CH2vfXXP/WQgJw+Qrn72umKs9UlGYQvn+QdnwO8Nc=", + "h1:J7L5bgyYNRAbtwAFJl2Lj+IMI2DJTrbbL33PTK4OWVY=", + "h1:JJ+EJQ+sIN3XRmNmrSUnUQtR8i3P22z+AbtAf8O/cRE=", + "h1:Wm5Ofhc15lX1OMMCt7iDV0NY5FDIouQDjX7I1iab55s=", + "h1:crKvBCgX6RlMcE6Ewm8o8YVuIg6mkXqKNgt/kSFYTvQ=", + "h1:zef23ac/YWw9O2FepFWRs+my9iWWUkniL4dT4LnCKjU=", + "zh:1a41f3ee26720fee7a9a0a361890632a1701b5dc1cf5355dc651ddbe115682ff", + "zh:30457f36690c19307921885cc5e72b9dbeba369445815903acd5c39ac0e41e7a", + "zh:42c22674d5f23f6309eaf3ac3a4f1f8b66b566c1efe1dcb0dd2fb30c17ce1f78", + "zh:4cc271c795ff8ce6479ec2d11a8ba65a0a9ed6331def6693f4b9dccb6e662838", + "zh:60932aa376bb8c87cd1971240063d9d38ba6a55502c867fdbb9f5361dc93d003", + "zh:864e42784bde77b18393ebfcc0104cea9123da5f4392e8a059789e296952eefa", + "zh:9750423138bb01ecaa5cec1a6691664f7783d301fb1628d3b64a231b6b564e0e", + "zh:e5d30c4dec271ef9d6fe09f48237ec6cfea1036848f835b4e47f274b48bda5a7", + "zh:e62bd314ae97b43d782e0841b13e68a3f8ec85cc762004f973ce5ce7b6cdbfd0", + "zh:ea851a3c072528a4445ac6236ba2ce58ffc99ec466019b0bd0e4adde63a248e4", + ] +} diff --git a/deploy/tofu/main.tf b/deploy/tofu/main.tf new file mode 100644 index 0000000..3f8f04c --- /dev/null +++ b/deploy/tofu/main.tf @@ -0,0 +1,322 @@ +# The parent instance that hosts the enclave. +# +# What this brings up is deliberately the *untrusted* half: a machine, a +# network path to it, and permission to read the buckets. The enclave's +# identity comes from PCR0, which is a property of the image rather than of +# anything declared here — so nothing in this file needs to be trusted for the +# attestation argument to hold, and none of it is a place to put a secret. +# +# nix build .#eif && packer build … deploy/ami # → ami-… +# tofu init && tofu apply -var ami_id=ami-… +# +# Storage is out of scope on purpose. The roots bucket carries Object Lock +# COMPLIANCE retention, which cannot be shortened or removed by anyone, +# including AWS support; creating that from a general-purpose deployment is a +# way to lose a bucket for ten years by typo. Point this at buckets that +# already exist. + +terraform { + required_version = ">= 1.6" + required_providers { + aws = { + source = "hashicorp/aws" + version = "~> 5.0" + } + } +} + +provider "aws" { + region = var.region +} + +locals { + name = "${var.name_prefix}-${var.environment}" + + tags = { + Name = local.name + Environment = var.environment + Component = "nitro-enclave-parent" + ManagedBy = "opentofu" + } +} + +# --------------------------------------------------------------------------- +# Network +# --------------------------------------------------------------------------- + +resource "aws_vpc" "this" { + cidr_block = var.vpc_cidr + enable_dns_support = true + enable_dns_hostnames = true + tags = local.tags +} + +resource "aws_internet_gateway" "this" { + vpc_id = aws_vpc.this.id + tags = local.tags +} + +resource "aws_subnet" "public" { + vpc_id = aws_vpc.this.id + cidr_block = var.subnet_cidr + availability_zone = var.availability_zone + map_public_ip_on_launch = true + tags = local.tags +} + +resource "aws_route_table" "public" { + vpc_id = aws_vpc.this.id + + route { + cidr_block = "0.0.0.0/0" + gateway_id = aws_internet_gateway.this.id + } + + tags = local.tags +} + +resource "aws_route_table_association" "public" { + subnet_id = aws_subnet.public.id + route_table_id = aws_route_table.public.id +} + +# --------------------------------------------------------------------------- +# Access +# --------------------------------------------------------------------------- + +resource "aws_security_group" "enclave" { + name = "${local.name}-enclave" + description = "Inbound HTTPS to the enclave; egress for S3, KMS and ACME" + vpc_id = aws_vpc.this.id + tags = local.tags + + # TLS terminates *inside* the enclave, so what passes through here is + # ciphertext the parent cannot read. This rule admits traffic to the machine; + # it does not grant the machine sight of it. + ingress { + description = "HTTPS, terminated inside the enclave" + from_port = 443 + to_port = 443 + protocol = "tcp" + cidr_blocks = var.ingress_cidrs + } + + # No SSH. Nothing here is meant to be reached by hand, and an open shell on + # the parent is the most useful thing an attacker could be given — it is the + # machine that proxies every byte in and out of the enclave. Use SSM if a + # session is genuinely needed. + egress { + description = "Outbound for S3, KMS and the ACME provider" + from_port = 0 + to_port = 0 + protocol = "-1" + cidr_blocks = ["0.0.0.0/0"] + } +} + +# --------------------------------------------------------------------------- +# Identity +# --------------------------------------------------------------------------- + +resource "aws_iam_role" "parent" { + name = "${local.name}-parent" + tags = local.tags + + assume_role_policy = jsonencode({ + Version = "2012-10-17" + Statement = [{ + Effect = "Allow" + Action = "sts:AssumeRole" + Principal = { Service = "ec2.amazonaws.com" } + }] + }) +} + +# The parent proxies the enclave's S3 traffic, so its role is what reaches the +# buckets. That is not a hole in the design: the objects are encrypted under +# keys derived from a master secret the parent does not hold, so this grants +# the ability to serve and delete ciphertext, not to read it. +data "aws_iam_policy_document" "buckets" { + statement { + sid = "ReadWriteObjects" + effect = "Allow" + actions = [ + "s3:GetObject", + "s3:PutObject", + "s3:DeleteObject", + "s3:ListBucket", + "s3:GetBucketLocation", + # Root records are read by version, underneath any delete marker. + "s3:ListBucketVersions", + "s3:GetObjectVersion", + # A PutObject that carries retention headers needs this too. + "s3:PutObjectRetention", + ] + resources = concat( + [for b in var.buckets : "arn:aws:s3:::${b}"], + [for b in var.buckets : "arn:aws:s3:::${b}/*"], + ) + } + + # The anchor chain is what makes rollback detectable, and Object Lock is what + # makes it immutable. Nothing on the parent has any business relaxing that, + # so the permission to do it is withheld rather than merely unused. + # PutObjectRetention is not in this list because the runtime needs it to + # write roots at all, and in COMPLIANCE mode it can only extend a retention, + # never shorten one. + statement { + sid = "NeverWeakenRetention" + effect = "Deny" + actions = [ + "s3:PutBucketObjectLockConfiguration", + "s3:PutObjectLegalHold", + "s3:BypassGovernanceRetention", + ] + resources = ["*"] + } +} + +resource "aws_iam_role_policy" "buckets" { + name = "${local.name}-buckets" + role = aws_iam_role.parent.id + policy = data.aws_iam_policy_document.buckets.json +} + +# --------------------------------------------------------------------------- +# Guest logs +# --------------------------------------------------------------------------- + +# For building log ARNs by hand — see the policy below. +data "aws_caller_identity" "current" {} + +# Created here, not by the enclave. The enclave holds `logs:PutLogEvents` and +# nothing more, so a compromised or misconfigured image cannot create log +# groups — and a typo in the group name fails its boot loudly instead of +# quietly filling a new group nobody is watching. +resource "aws_cloudwatch_log_group" "guest" { + name = "/${local.name}/guest" + + # Set deliberately. Left unset these never expire, and the contents are + # attacker-controlled: whatever a guest chose to write to stdout. An + # unbounded retention on unbounded input is an unbounded bill. + retention_in_days = var.guest_log_retention_days + + tags = local.tags +} + +resource "aws_cloudwatch_log_stream" "guest" { + name = "guest" + log_group_name = aws_cloudwatch_log_group.guest.name +} + +# This is the **parent instance's** role, and it is the identity that will make +# the call. The enclave is not an IAM principal and has none of its own: it has +# no NIC, so its SDK reaches IMDS through gvproxy on this instance and receives +# these credentials. When the enclave gets an attested identity of its own, +# this statement moves there and the parent stops being able to write to the +# stream at all. +data "aws_iam_policy_document" "guest_logs" { + statement { + sid = "WriteGuestLogs" + effect = "Allow" + actions = ["logs:PutLogEvents"] + # The one stream, not the group and not `*`. Nothing here needs to create + # a group, a stream, or write to anyone else's. + # + # Built from components rather than from `aws_cloudwatch_log_group.arn`, + # which is inconsistent about carrying a trailing `:*`. Appending to it + # would silently produce an ARN matching nothing — and the symptom is an + # `AccessDeniedException`, which this runtime treats as a deployment + # mistake and refuses to boot on. Worth the extra interpolation. + resources = [ + "arn:aws:logs:${var.region}:${data.aws_caller_identity.current.account_id}:log-group:${aws_cloudwatch_log_group.guest.name}:log-stream:${aws_cloudwatch_log_stream.guest.name}", + ] + } +} + +resource "aws_iam_role_policy" "guest_logs" { + name = "${local.name}-guest-logs" + role = aws_iam_role.parent.id + policy = data.aws_iam_policy_document.guest_logs.json +} + +# For a shell without opening port 22. +resource "aws_iam_role_policy_attachment" "ssm" { + role = aws_iam_role.parent.name + policy_arn = "arn:aws:iam::aws:policy/AmazonSSMManagedInstanceCore" +} + +resource "aws_iam_instance_profile" "parent" { + name = "${local.name}-parent" + role = aws_iam_role.parent.name + tags = local.tags +} + +# --------------------------------------------------------------------------- +# The instance +# --------------------------------------------------------------------------- + +resource "aws_instance" "parent" { + ami = var.ami_id + instance_type = var.instance_type + subnet_id = aws_subnet.public.id + vpc_security_group_ids = [aws_security_group.enclave.id] + iam_instance_profile = aws_iam_instance_profile.parent.name + + # The whole point of the machine. Without it `nitro-cli run-enclave` fails + # with a message about the driver rather than about this flag. + enclave_options { + enabled = true + } + + # Nitro Enclaves needs at least 4 vCPUs, because the allocator reserves whole + # cores for the enclave and the parent still has to run. Smaller types are + # accepted by the API and then cannot start an enclave. + lifecycle { + precondition { + condition = can(regex("\\.(x|2x|4x|8x|12x|16x|24x|metal)large$", var.instance_type)) || can(regex("\\.xlarge$", var.instance_type)) + error_message = "instance_type must be enclave-capable with >= 4 vCPUs (xlarge or bigger)." + } + } + + root_block_device { + volume_size = 32 + volume_type = "gp3" + encrypted = true + } + + metadata_options { + http_tokens = "required" # IMDSv2 + http_endpoint = "enabled" + + # UNVERIFIED PRODUCTION DEPENDENCY. Nothing in this repository proves that + # an enclave can actually reach IMDS: the QEMU harness has no instance + # metadata service to reach, so gvproxy's handling of 169.254.169.254 has + # never been exercised. Every AWS call the enclave makes — S3, KMS, SSM and + # now CloudWatch — rests on it. Validate IMDSv2 credential resolution on + # Nitro hardware before production; if gvproxy routes rather than proxies, + # this must be 2, and the symptom is calls failing in a way that reads like + # missing credentials rather than like a network fault. + # + # One hop is right *if* gvproxy proxies — it terminates the enclave's + # connection here and opens its own, so the request originates on this + # instance. Stated rather than defaulted, because the value is a deployment + # decision and not an incidental. + http_put_response_hop_limit = 1 + } + + # Configuration the image cannot know: which buckets, and which domain the + # certificate is for. Not secrets — the master key is not passed here, and + # once M8 lands it comes from KMS gated on PCR0 rather than from anywhere on + # this machine. + user_data = templatefile("${path.module}/user-data.sh.tftpl", { + data_bucket = var.buckets[0] + roots_bucket = length(var.buckets) > 1 ? var.buckets[1] : var.buckets[0] + region = var.region + tls_domains = join(",", var.tls_domains) + }) + + user_data_replace_on_change = true + + tags = local.tags +} diff --git a/deploy/tofu/outputs.tf b/deploy/tofu/outputs.tf new file mode 100644 index 0000000..a1179e4 --- /dev/null +++ b/deploy/tofu/outputs.tf @@ -0,0 +1,38 @@ +output "public_ip" { + value = aws_instance.parent.public_ip + description = "Where the enclave answers on :443" +} + +output "instance_id" { + value = aws_instance.parent.id +} + +# What to run once it is up. The last command is the one that matters: it +# checks that the certificate the connection was served is the one the enclave +# attested, and that the runtime and guest behind it are the approved ones. +output "verify" { + value = <<-EOT + # Before the first start, and for every guest change: upload the guest the + # key policy pins. The key must match `guestObject` in + # deploy/nix/deployment.nix. The enclave measures what it fetches into + # PCR16 and asks KMS for its key with that measurement, so an object the + # policy does not name boots an enclave that can read nothing. + nix build .#guest-release --out-link guest-release + aws s3 cp guest-release/guest.wasm \ + s3://${length(var.buckets) > 1 ? var.buckets[1] : var.buckets[0]}/guest/guest.wasm + + # The key policy's condition, on both kms:GenerateDataKey and kms:Decrypt. + # Replace PCR16 on a guest change; never add a second value beside it. + # "kms:RecipientAttestation:PCR0": "$(jq -r .PCR0 result/pcr.json)" + # "kms:RecipientAttestation:PCR16": "$(jq -r .PCR16 guest-release/guest-pcr16.json)" + + # Watch it come up (no SSH port is open; this uses SSM) + aws ssm start-session --target ${aws_instance.parent.id} + journalctl -u enclave.service -f + + # Verify from anywhere + nitro-attest --url https://${aws_instance.parent.public_ip}/auth/ \ + --pcr0 "$(jq -r .PCR0 result/pcr.json)" \ + --guest guest-release/guest.wasm + EOT +} diff --git a/deploy/tofu/user-data.sh.tftpl b/deploy/tofu/user-data.sh.tftpl new file mode 100644 index 0000000..7cd02da --- /dev/null +++ b/deploy/tofu/user-data.sh.tftpl @@ -0,0 +1,22 @@ +#!/usr/bin/env bash +# Deployment configuration the image cannot know: which buckets, which region, +# which domain. Written where the enclave.service unit picks it up. +# +# Not a place for secrets. This file is readable from the instance metadata +# service by anything on the box, and the parent is the party the enclave +# excludes. The master key is deliberately absent — until M8 it is supplied +# out of band, and after M8 it comes from KMS gated on PCR0, which is a +# statement about the enclave's code rather than about this machine. +set -euxo pipefail + +mkdir -p /etc/systemd/system/enclave.service.d +cat > /etc/systemd/system/enclave.service.d/deployment.conf <= 1 && length(var.buckets) <= 2 + error_message = "Give one bucket, or two as [data, roots]." + } +} + +variable "tls_domains" { + type = list(string) + default = [] + description = "Domains for the enclave's certificate. Required for ACME; a self-signed certificate needs none, since attestation rather than a CA is what a client checks." +} + +variable "vpc_cidr" { + type = string + default = "10.42.0.0/16" +} + +variable "subnet_cidr" { + type = string + default = "10.42.1.0/24" +} + +variable "availability_zone" { + type = string + default = null + description = "Defaults to the first AZ in the region." +} + +variable "ingress_cidrs" { + type = list(string) + default = ["0.0.0.0/0"] + description = "Who may reach :443. Open by default because the endpoint is meant to be public and its TLS terminates inside the enclave." +} + +variable "guest_log_retention_days" { + description = "How long to keep guest stdout/stderr. The contents are chosen by the guest, so this is a cost bound as much as a policy." + type = number + default = 30 +} diff --git a/docs/BACKGROUND_TASKS.md b/docs/BACKGROUND_TASKS.md new file mode 100644 index 0000000..491b22f --- /dev/null +++ b/docs/BACKGROUND_TASKS.md @@ -0,0 +1,173 @@ +# Background tasks + +The runtime schedules durable work for individual tenants inside a running +enclave. A tenant with no due work needs no running guest instance. One active +enclave owns the queue; running multiple schedulers over the same filesystem +requires an ownership/fencing protocol that is not implemented here. + +## Enable + +The guest must export `run-task` and may import the queue interface defined in +[`wit/tasks/tasks.wit`](../wit/tasks/tasks.wit). The HTTP example implements both +its usual `wasi:http` interface and this background interface. + +Enable `S3FS_BACKGROUND_TASKS=true` (or `--background-tasks true`) alongside +WebAuthn authentication and tenant isolation. In the Nix deployment, set +`backgroundTasks = true` in `deploy/nix/deployment.nix`. This image configuration +change changes PCR0; rebuilding the example guest changes PCR16. Update the +approved measurements through your normal deployment process. + +The emulator image turns it on independently, in `eif-qemu`'s environment in +`flake.nix`, so the end-to-end harness can exercise this path. That override +changes only the emulator's PCR0 — which the harness checks against its own +build rather than a published number — and leaves the production default alone. + +| Setting | Default | Purpose | +|---|---:|---| +| `S3FS_BACKGROUND_CONCURRENCY` | 1 | Maximum active background workers | +| `S3FS_BACKGROUND_TIMEOUT_SECS` | 30 | Maximum duration of one guest attempt | +| `S3FS_BACKGROUND_MAX_RECORDS` | 1024 | Total durable records, including terminal tasks | +| `S3FS_BACKGROUND_PER_TENANT` | 64 | Records allowed per tenant | + +All limits must be positive. Background workers have separate admission from +interactive requests and yield on epoch ticks. They use the same tenant lock, +so one tenant's background and interactive code cannot run simultaneously. +A busy tenant is deferred without spending an execution attempt. Due tenants +take turns, keeping deadline order within each tenant, so one tenant’s backlog +cannot monopolize all execution slots. These limits bound background work; they are not a global limit on interactive traffic or a +reservation of physical CPU cores. Guest linear memory is unrestricted for now. +There is no GPU task execution path. + +## Guest contract + +During an authenticated interactive call, the guest can call: + +```text +enqueue(id, payload, run-at, interval-ms) -> result +status(id) -> result +cancel(id) -> result +forget(id) -> result +``` + +The runtime supplies the tenant identity from the executing instance. None of +these functions accepts a tenant ID. Anonymous calls and guest initializers +have no queue authority. Queue state lives at `/runtime/tasks` in the shared +encrypted filesystem, above every guest's filesystem scope. + +An ID is a tenant-local idempotency key: 1–64 ASCII letters, digits, hyphens or +underscores. Repeating an enqueue with the same ID, payload, requested time and +interval returns the original task. Different input under that ID is refused. +The payload and result each have a 64 KiB limit. Task status includes the state, +attempt count, occurrence, result bytes and a bounded error message. Result +bytes are represented as a JSON array of byte values. + +`run-at` is Unix time in milliseconds; zero means immediately. An optional +interval creates a recurring schedule and must be at least 1000 milliseconds. +The guest decides what the payload means and which operations its user may +schedule. A recurring polling task can inspect that tenant's own request files +in each callback. Use enqueue's payload as the durable request when possible: +writing a separate application file and enqueueing are **not** one transaction. + +The runtime invokes this component export when the task is due: + +```text +run-task(task-id, payload) -> result +``` + +This export is not an HTTP endpoint. It receives the correct tenant's directory +as `/` and an execution deadline. It uses a fresh guest +instance under the same pool lock as HTTP calls; any warm HTTP instance is +dropped first so it cannot retain stale database handles across the mutation. +A missing tenant directory causes failure, never recreation. + +The background callback can inspect task status, but cannot enqueue, cancel or +forget tasks. An authenticated interaction grants standing authorization to run +the submitted job later; the original short-lived interaction token is not +replayed. Revoking a passkey does not automatically cancel a tenant's schedules: +use cancellation to revoke that standing job authorization. The approved guest +must validate task types and payloads before enqueueing them. Jobs execute the +currently approved component after an upgrade, so keep payloads compatible or +cancel affected jobs before deploying an incompatible guest. + +## Durability and retries + +Each full record is written to a temporary file, committed, then atomically +renamed into place. The durable record is the scheduling index. There is no +second notification write whose loss could strand an acknowledged task. +Startup reconstructs the in-memory deadline queue from records and discards +unpublished temporary files; corrupt published records stop startup. + +Delivery is **at least once**. An execution intent is persisted before calling +the guest, and the result is persisted after the callback returns. A crash +between an effect and its completion record can repeat the callback. The `task-id` +passed to `run-task` is this run ID, **not** the ID the task was enqueued under: + +```text +:: e.g. nightly-sync:9f86d081884c7d659a2feaa0c55ad015:3 +``` + +`generation` is 32 hex digits minted at enqueue and `occurrence` counts a +recurring task's runs from 0. Because an ID cannot contain `:`, the enqueued ID +is everything before the first `:` — a guest that checks `task-id` against the +ID it enqueued will refuse every run. The run ID stays the same across retries, +and changes for a later recurring occurrence or a new task after `forget`. +Deduplicate effects by the whole run ID. Commit guest filesystem writes before +returning success. External effects need their own idempotency mechanism. + +Failed attempts retry with exponential backoff, up to five attempts per +occurrence. A recovered running attempt consumes its existing attempt count. +Exhausted tasks become `failed`; recurring schedules stop on terminal failure. +Recurring successes schedule the next occurrence at a stable tenant/job phase. +Missed intervals coalesce into one check rather than being replayed. Overdue +records receive up to one second of recovery jitter, and concurrency remains +bounded even if all tasks are due. + +Cancellation prevents future attempts and wins over a late completion record. +It does not undo side effects or forcibly interrupt an already executing +callback. A client's cancellation request may wait for the tenant's running +callback to release the shared lock. `forget` removes a terminal record to free +quota, and refuses records still owned by a worker. Terminal records are not +automatically deleted. + +Scheduler storage failures stop the server rather than silently dropping work. +Stopping the enclave stops execution; the scheduler does not start the enclave +itself. Restart recovery shares the filesystem's existing rollback/freshness +assumptions; this queue does not add an external freshness witness. + +## HTTP example + +The example guest exposes these authenticated application routes: + +| Request | Behavior | +|---|---| +| `POST /tasks/` | Enqueue the body as the task payload; returns 202 | +| `GET /tasks/` | Read that tenant's task record | +| `DELETE /tasks/` | Cancel the task | +| `DELETE /tasks//forget` | Remove a terminal task record | + +Optional `x-task-run-at` and `x-task-interval-ms` headers set the requested time +and interval. The callback writes one durable result file per run ID inside the +tenant's `/http-example/tasks/` directory, returning an existing result on retry. +Its `fail`, `spin` and `check-authority` payloads exercise retries, deadlines and +the prohibition on background code authorizing more work. + +Build and test: + +```sh +cargo build --manifest-path examples/guest-http/Cargo.toml --release --target wasm32-wasip2 +# One serve_auth test drives the real client binary as a subprocess. Without it +# that test skips with a notice instead of failing, which reports green for a +# test that never ran. +cargo build --release -p enclave-runtime --features testing --bin passkey-client +cargo test -p enclave-runtime --lib tasks::tests -- --include-ignored +# The whole suite, deliberately unfiltered: a name filter here silently stopped +# covering this feature once a second scheduling test was added under a +# different name. +cargo test -p enclave-runtime --test serve_auth -- --include-ignored +``` + +Inside an enclave, step `6/8` of [`deploy/qemu-nitro/run-e2e.sh`](../deploy/qemu-nitro/run-e2e.sh) +schedules work through the real passkey client and waits for it to finish. That +is the only place this path runs against a real scheduler, a mounted filesystem +and a certificate the enclave obtained itself — everything above it is in +process, where the standing-authorization claim is made by a stand-in. diff --git a/docs/CLIENT_INTEGRATION.md b/docs/CLIENT_INTEGRATION.md new file mode 100644 index 0000000..e7397c1 --- /dev/null +++ b/docs/CLIENT_INTEGRATION.md @@ -0,0 +1,285 @@ +# Integrating a client against this runtime + +For a team with an existing Nitro client, pointing an app at this enclave for +the first time. + +If you have been talking to a **nitriding** enclave, the short version is: the +endpoints, the `user_data` layout and the trust model are all different, and +there is a per-request passkey assertion that nitriding has no equivalent of. +This is not a base-URL change. It is maybe a day of client work, and most of it +is in one file. + +You can have the enclave running locally, serving your own guest, in about ten +minutes — see [Run it](#run-it) — and develop against the same verification code +path you will run in production. + +## What is different, in one table + +| | nitriding | this runtime | +|---|---|---| +| get the document | `GET /enclave/attestation?nonce=` | `x-enclave-attestation` header on any `/auth/*` response | +| enclave info | `GET /v1/enclave-info` | none; everything is in the document | +| `user_data` | `"sha256:" + tlsKeyHash + ";" + "sha256:" + appKeyHash` | 68 bytes: `0x1220 ‖ sha256(tls_cert_der) ‖ 0x1220 ‖ sha256(guest_component)` | +| pinned | PCR0 | PCR0 **and** PCR16 | +| per-request auth | none | a WebAuthn assertion bound to that exact method and path | +| tenancy | yours to build | one tenant per passkey, enforced by the runtime | + +The last two rows are the ones that change the app, not just the crate. + +## Run it + +Needs Linux with KVM, Docker and Nix. Once per boot: + +```bash +sudo modprobe vsock_loopback +``` + +Once ever: + +```bash +docker build -t s3fs-qemu-nitro:latest deploy/qemu-nitro +cargo install vhost-device-vsock --root target/qemu-nitro/tools +``` + +Then, with your component: + +```bash +cargo build --release --target wasm32-wasip2 # in your guest's crate +deploy/qemu-nitro/dev-enclave.sh --guest target/wasm32-wasip2/release/cosigner.wasm +``` + +It builds the enclave image, starts a block store and an ACME CA, boots the +enclave under QEMU's `nitro-enclave` machine, waits for it to fetch and measure +your component and obtain a certificate, then prints the three values a client +pins and stays up. Ctrl-C stops everything it started. + +There is nothing to configure. The guest is fetched from the store at boot and +measured into PCR16 before it can obtain a key, so a new build is a new PCR16 +and a restart — see [dev-enclave](DEV_ENCLAVE.md) for the details, the options +and what is and is not real about it. + +## The protocol + +Only `/auth/*` answers without an assertion. Everything else reaches the guest, +and nothing reaches the guest without a fresh approval. + +### Registering + +Registration is **open**: any passkey may enrol, and each one gets its own +tenant with its own isolated filesystem. There is no invitation, no enrolment +token and no operator step. + +``` +POST /auth/register/options {"display_name": "optional"} + → 200 {"registration_id": "...", "options": } + +POST /auth/register/verify {"registration_id": "...", "credential": } + → 200 {"tenant_id": "...", "credential_id": "..."} +``` + +`options` is what you hand to `navigator.credentials.create()` — or to your +platform authenticator. Keep `credential_id`; every later request names it. + +### One request + +Four steps, because the approval is bound to the request rather than to a +session. + +``` +1. POST /auth/request/options + {"credential_id": "...", "method": "POST", "path": "/sign", "query": null} + → 200 {"challenge_id": "...", "options": } + + header x-enclave-attestation: + +2. verify that document (next section) before doing anything with it + +3. sign the challenge → POST /auth/request/verify + {"challenge_id": "...", "assertion": ""} + → 200 {"token": "...", "expires_in_secs": 60} + +4. the real request, with Authorization: Bearer +``` + +Notes that matter: + +- The challenge is issued **for that method, path and query**. A token moved to + another route is refused — this is enforced, not advisory. +- A token is **single use** and expires in 60 seconds. That bounds the time to + *start* an interaction, not how long one may run: a long stream approved once + keeps going. +- `deny_unknown_fields` is set on both request bodies. An extra field is a 400, + deliberately — this route used to take a `body_sha256` and a client still + sending one must be told it means nothing now rather than have it ignored. +- The assertion is one header holding the credential JSON verbatim, not five + holding its parts. +- Auth headers are stripped before the guest sees the request. A guest can + neither read a client's token nor forge one. + +### Step 2, in full + +This is the part worth implementing carefully; it is what the whole design rests +on. Given the document bytes from `x-enclave-attestation`: + +1. **Signature and chain.** COSE_Sign1, ES384. `cabundle` is ordered root + first; the chain to check is `cabundle` in order, then `certificate`. Every + certificate except the leaf must be a CA, each must be validly issued by the + one before it, and each must be inside its validity window. +2. **Root.** The presented root must equal the one you pinned. Do not accept + "some root" — see [Dev and production](#dev-and-production). +3. **PCR0**, against the value the image build published. +4. **PCR16**, against the value your component measures to. `nitro-attest + --measure .wasm` prints it, and the runtime computes it the same + way. +5. **`user_data` binds this connection.** 68 bytes, two sha2-256 multihashes: + + ``` + 0x12 0x20 ‖ sha256(tls_certificate_der) ‖ 0x12 0x20 ‖ sha256(guest_component) + ``` + + The first must equal sha256 of the DER of the certificate **this TLS + connection was served**. That is what ties the document to the channel you + are on; without it, a verified document proves an enclave exists somewhere, + not that you are talking to it. +6. **Nonce** — the one you sent, echoed back. It is not optional: every request + must carry `x-enclave-nonce` (unpadded base64url, 8–64 bytes), and one + without it is refused with a 400 before routing, with no document. +7. **Age.** The document carries the enclave's own timestamp. Reject an old one. + +Why both registers, since it is the usual question: PCR0 is measured by the +hypervisor from the image, so nothing inside the enclave can choose it — but the +image does not contain the guest. PCR16 is the guest, measured and locked by the +runtime before it could obtain a key — but because the *runtime* writes it, an +enclave running somebody else's runtime could claim your guest's value. PCR0 +says the runtime that wrote PCR16 is yours; PCR16 says which application it +loaded. Neither substitutes for the other. + +Guest responses deliberately carry no document. By then you have pinned the +certificate, and TLS proves the peer still holds its key — so **refuse any +connection that serves a different certificate**, and attest again before using +it. A certificate renewal looks exactly like that, and so does an interception; +a fresh document is what tells them apart. + +### Attesting without a request + +Every response under `/auth/` carries a document, whatever its status. To +attest the enclave without asking it for anything — at startup, or after the +certificate changed — send `GET /auth/` with a nonce: the answer is a 405, and +the document on it is the point. `nitro-attest --url https://` does exactly +this. + +### What the guest sees + +Your component receives an ordinary `wasi:http` request with +`x-enclave-tenant` set to the caller's tenant, and a filesystem rooted at that +tenant's own directory. Path traversal out of it is refused by the runtime, not +by the guest. You do not have to implement tenant separation; you have to not +work around it. + +## What changes in an existing verifier + +A crate shaped like a pure-verification library — bytes in, no HTTP, no async — +is exactly the right shape. The changes are: + +1. **Source of the document**: the `x-enclave-attestation` response header on + `/auth/*`, not a GET endpoint. +2. **Add a trust-root parameter.** Default to the AWS Nitro root; allow an + override. This is the only thing that differs between dev and production. +3. **Add PCR16** to what is pinned, alongside PCR0. +4. **Replace the `user_data` parser** with the multihash layout above, and + actually compare against the served certificate — which means the HTTP layer + has to hand the peer certificate down to the verifier. +5. **App side**: the options → assertion → verify → token sequence, and storing + `credential_id` from registration. + +### One decision to make + +This repo's [`nitro-attestation`](../crates/nitro-attestation) crate already +does all of step 2–6 above, has no dependency on any enclave-side code by +design, and is what the runtime and every end-to-end leg are tested against. +Depending on it directly means one verifier instead of two implementations that +have to agree. + +The catch is real, though: it verifies with `aws-lc-rs`, which needs a C +toolchain and CMake to cross-compile. A pure-Rust `p384`/`x509` stack is +friendlier for iOS and Android builds. So this is a trade-off — one maintained +verifier against an easier mobile build — and not an obvious win either way. + +If you keep your own, the two things most worth copying are the chain order +(`cabundle` root-first, then the leaf) and the rule that a non-pinned root is +reported distinctly rather than accepted quietly. + +## Dev and production + +The difference is one value. + +| | dev enclave | production | +|---|---|---| +| trust root | the file `dev-enclave.sh` prints | omit — the AWS Nitro root | +| PCR0 | printed at startup | from the release image build | +| PCR16 | printed at startup | from your component's release build | +| CA | Pebble's root, also printed | a public root | + +Everything else — the endpoints, the verification steps, the failure modes — is +identical. That is the point: you are not developing against a relaxed path and +meeting the real one in production. + +**Do not set an "allow untrusted root" flag to make dev work.** A verifier that +accepts whatever root a document arrived with is checking nothing, and it will +pass in production too. Pin the dev root as a root; the code path is then the +same one production uses. + +## What the local enclave does not prove + +Worth being plain about, because it is easy to assume otherwise. + +QEMU's emulated NSM does not sign anything — its source says *"we don't actually +sign the data, so we use -1 as the 'alg' value"*, and -1 is not a COSE algorithm +identifier. So the emulator image mints a certificate chain at boot and re-signs +the documents the device produced, contents untouched. Your client then runs its +whole verification path against something it can accept. + +But the key is inside an image whoever runs it controls. A verified document +locally means *"this image said so"*, where on hardware it means *"a Nitro +enclave with this measurement said so"*. The root changes at every boot for +exactly that reason: there is deliberately nothing to hardcode into an app. + +KMS is not exercised either — it will not release a key against a document it +cannot trace to a Nitro root, so the local image uses a static master key. On +hardware, a key policy pinning PCR0 and PCR16 is what stops a substituted guest +from reading anything; locally, a substituted guest boots and serves, and only +the client's PCR16 check refuses it. + +Neither limitation affects the client code you write. Both are reasons to run +against real hardware before you ship. + +## Settings the local enclave uses + +| | | +|---|---| +| URL | `https://127.0.0.1:8443` (override with `--port`) | +| certificate name | `enclave.test`, issued by Pebble over real ACME | +| WebAuthn RP id | `enclave.test` | +| WebAuthn origin | `https://enclave.test` — claimed in the assertion, even though you connect to `127.0.0.1` | +| challenge TTL | 60s | +| token TTL | 60s, single use | + +The origin is the one that trips people up: it is compared exactly, not by +suffix, so an assertion claiming `https://127.0.0.1:8443` is refused. Production +needs the RP id to be a domain your app is actually associated with. + +## Reference client + +[`runtime/src/bin/passkey-client.rs`](../runtime/src/bin/passkey-client.rs) +implements all of the above, including a software authenticator, and is the +executable answer when something here disagrees with it — it is what the end-to-end +harness drives, so it is known to work against a running enclave. + +```bash +passkey-client --url https://127.0.0.1:8443 --state alice.json \ + --trust-root --pcr0 --pcr16 \ + get --path /whatever-your-guest-serves +``` + +`--dump-proof ` writes the three things a verifier cannot recover +afterwards — the document, the certificate that connection was served, and the +nonce — which is the fastest way to test a verifier offline. diff --git a/docs/COMPATIBILITY.md b/docs/COMPATIBILITY.md index 67e7726..d25e13c 100644 --- a/docs/COMPATIBILITY.md +++ b/docs/COMPATIBILITY.md @@ -1,64 +1,107 @@ # Compatibility Matrix -The user-facing semantic contract: what `wasi:filesystem@0.2.x` ops do behind our implementation, and where they deviate from POSIX. +What a Wasm guest gets from `wasi:filesystem@0.2.6`, and where it differs from a local filesystem. -S3 is an object store, not a filesystem. Some POSIX semantics map cleanly, some require workarounds, some are fundamentally incompatible. This document is honest about all three categories. +The storage engine is a ZFS-style copy-on-write block store: content is held in immutable, AEAD-encrypted blocks packed into slab objects, and the whole filesystem hangs off one signed, hash-chained root record. Every mutation is a single transaction group and therefore a single root record, which is where most of the guarantees below come from. ---- - -## ✅ Fully POSIX-equivalent +## ✅ POSIX-equivalent | WASI op | Notes | |---|---| -| `read`, `read-via-stream` | Ranged `GetObject` with buffer-pool cache. Streams are non-blocking; `Pollable::ready()` does the actual fetch. | -| `write`, `write-via-stream`, `append-via-stream` | `MultipartUpload` + `UploadPartCopy` for in-place updates of large files (GeeseFS-parity); single `PutObject` for small files. Streams queue + drain via `Pollable::ready()`. | -| `stat`, `stat-at` | Race-three lookup on cache miss (parallel HEAD + HEAD-with-slash + LIST). `DescriptorStat` returns mtime from `x-amz-meta-s3wasifs-mtime` if set, else S3's `LastModified`. | -| `read-directory` | Snapshot at iterator open via paginated `ListObjectsV2(prefix, delim='/')`. Per WASI semantics, entries added/removed mid-iteration may be missed (POSIX-allowed). | -| `create-directory-at` | Zero-byte `dir/` marker object. | -| `is-same-object` | Inode-ID comparison; correct **within one mount**. | -| `get-flags`, `get-type` | Trivial. `get-type` returns `regular-file` / `directory` / `symbolic-link`. | -| `metadata-hash{,-at}` | `siphash24(etag ‖ size)` for files; siphash of sorted child entries for directories. Changes when content / membership changes. | -| `advise` (`will-need`, `sequential`) | Honoured as buffer-pool prefetch hints. Other variants are no-ops (allowed by WASI). | -| `open-at` with `create-flags::create \| exclusive` (`O_EXCL`) | Atomic via S3 `PutObject` with `If-None-Match: *`. Conflict → `error-code::exist`. | -| `open-at` with `open-flags::truncate` | Synchronous zero-byte `PutObject` on open, regardless of whether the guest later writes. | -| `open-at` with `path-flags::symlink-follow` set | Recursive resolution, max-depth 40 (Linux `MAXSYMLINKS`). Cycles → `error-code::loop`. | -| `open-at` with `path-flags::symlink-follow` cleared on a symlink | Returns `error-code::loop` per WASI spec (POSIX `O_NOFOLLOW`). | -| `symlink-at` | Atomic via `If-None-Match: *`; conflict → `error-code::exist`. Stored as a small object with body = target string and `x-amz-meta-s3wasifs-type=symlink`. | -| `readlink-at` | Single `GetObject`; non-symlink target → `error-code::invalid` per POSIX. | -| `unlink-file-at` | `DeleteObject`. Inode marked `Deleted`; further ops on stale handles return `bad-descriptor`/`no-entry`. | -| `remove-directory-at` | List with `MaxKeys=2` for emptiness check, then `DeleteObject` of marker if present. Non-empty → `error-code::not-empty` (POSIX). | -| `rename-at` (file or symlink) | **Async.** Returns as soon as the inode tree is rewired; `CopyObject` + `DeleteObject` runs on a background worker. During the in-flight window, reads/writes against the new path resolve to the OLD S3 key so the object never disappears. Worker errors surface on the next `sync` of the renamed inode. | -| `rename-at` (directory) | **Async.** Returns immediately after the inode tree is rewired. The worker paginates `ListObjectsV2` over the source prefix and copies + deletes each key in the background. Empty destination is replaced; non-empty → `not-empty`; rename-into-self → `invalid`. | -| `set-size` | **Shrink:** flush pending writes, GET `[0..new_size)`, single `PutObject`. **Grow:** zero-fill via buffer pool (capped at 100 MiB to avoid OOM). | -| `set-times{,-at}` | Persists as `x-amz-meta-s3wasifs-{atime,mtime}` via `CopyObject` self-copy with `MetadataDirective=REPLACE`. Read back into `DescriptorStat` on next lookup. | -| `sync`, `sync-data` | Drains any in-flight eager `UploadPart`s parked on parts, then for remaining dirty parts: `copy_unmodified_parts` → `CompleteMultipartUpload`. Falls back to single `PutObject` for sub-part / small-file paths. Durable when the call returns. Serialized against an in-flight rename worker via a per-inode async lock. | -| `preopens.get-directories` | Single entry: `(root_descriptor, mount_path)` where `mount_path` defaults to `/`. | - -## ⚠️ Weakened semantics — guests will mostly not notice, but the gap is real - -| WASI op | Gap | Mitigation / contract | +| `read`, `write`, `read-via-stream`, `write-via-stream`, `append-via-stream` | Buffered per handle at record granularity. | +| `sync`, `sync-data` | Commits a transaction group and publishes a new signed root. | +| `stat`, `stat-at` | Read from the dnode on every call, so they cannot go stale. `link-count` and all three timestamps are real values, not placeholders. | +| `set-times`, `set-times-at` | A field write in the dnode. No `CopyObject`, and nothing else to disagree with. | +| `set-size` | Truncate or grow, at any size. Growth is a hole and costs nothing. | +| `create-directory-at`, `remove-directory-at`, `unlink-file-at` | One commit each. | +| **`rename-at`** | **Atomic.** One directory-entry move inside one commit. A directory rename costs the same as a file rename. | +| `read-directory` | Name-ordered, straight from the directory's leaves — no sort step. | +| `symlink-at`, `readlink-at` | Targets up to 294 bytes live in the dnode; longer ones spill to data blocks. Cycle detection at depth 40. | +| `open-at` with `CREATE` / `EXCLUSIVE` / `TRUNCATE` | Create is atomic inside its transaction. | +| **`link-at`** | **Hard links.** A directory entry is an object id, so a link is one more entry and an increment of `nlink`. Files and symlinks only. | +| **`unlink-file-at` while an fd is open** | **POSIX behaviour.** The object stays readable and writable through existing handles until the last one closes. | +| `is-same-object`, `metadata-hash`, `metadata-hash-at` | Exact. Identity is `(object id, generation)`, which is stable across mounts because object ids are never reused. | + +**Crash consistency.** The visible state is always a Merkle-verified snapshot, never a torn one. Slabs written by a commit that died before publishing its root are orphans: unreferenced, unreachable, harmless. There is no journal and no fsck. + +**Integrity.** Every block read verifies a BLAKE3 checksum against its parent's block pointer and then opens an AEAD whose additional data binds the block to its exact position in the tree. A block cannot be corrupted, forged, substituted, or relocated without the read failing. + +**Rollback.** Root records are written with `If-None-Match: *` into a bucket under Object Lock COMPLIANCE retention, and each carries the hash of its predecessor. An adversary with full write access to the buckets can make the filesystem unreadable, but cannot make it read *wrong* and cannot make it read *old*. + +## ⚠️ Weakened semantics + +| WASI op | Gap | Contract | |---|---|---| -| `rename-at` | **Not POSIX-atomic.** Async `CopyObject` + `DeleteObject` is two calls; both keys briefly exist; a host crash mid-window leaves both. A second `rename` on the same inode while one is in flight returns `WouldBlock`. | Caller can `Fs::wait_for_rename` to drain. Errors stash on the inode and surface on the next `sync` / `rename`. Pre-flush of any open writable handle for the source happens before enqueue, so the worker copies a fully-committed object. | -| `unlink-file-at` while fd is open | **No POSIX "invisible deleted file" semantics.** Further ops on stale handles return `bad-descriptor`/`no-entry`. | Documented; matches GeeseFS. | -| Two writers to same key | **No fencing.** Last `MultipartUpload` committed wins; no torn data but no single-writer guarantee. | Inside an enclave there's typically one writer. If you need single-writer-per-key semantics, layer it above (DynamoDB CAS, lease service, etc.). | -| `is-same-object` across mounts | Inode IDs are per-process. Two mounts of the same bucket disagree on identity. | Within-mount works. | -| `set-times` and `LastModified` | We persist the user-set mtime as metadata, but every successful `CompleteMultipartUpload` / `PutObject` also bumps S3's own `LastModified`. The metadata mtime is what we report from `DescriptorStat`. | Acceptable for nearly all guests. | -| `set-size` grow > 100 MiB | Returns `error-code::invalid`. | Use `pwrite` of real bytes for genuine large grows. | -| `sync` of a sub-part write after MPU started | Currently triggers an MPU+`UploadPartCopy` rewrite path. Correct but slow when the dirty content is far smaller than `single_part_threshold`. | Performance gap, not correctness. | +| Crash while a file is unlinked-but-open | The dnode leaks: allocated, nameless, unreachable. | Garbage, not corruption — the same trade a real filesystem makes with its orphan inode list. Reclaimed by the garbage collector when that lands. | +| Two writers to the same file | Each handle buffers independently; whichever syncs last wins. | Within one mount, use one handle per file. | +| Two mounts of the same buckets | Exactly one can win a given root sequence. The loser is **poisoned**: every subsequent operation fails, including reads. | Deliberate. Retrying would reuse a transaction group and repeat every AEAD nonce in it. Single writer per filesystem. | +| Unsynced writes on a crash | Lost. | The transaction-group model: durability is at `sync` / `close`, and what survives is always consistent. | +| Cold-mount freshness | A store that *hides* roots newer than the one it serves is not cryptographically excluded. Object Lock means those roots cannot be deleted, so this requires S3 itself to lie. | Pass `--min-root-seq` to set a floor from outside the store. Within a session the gap does not exist: the accepted sequence only ever rises. | +| `atime` | Only updated when a handle is synced. | No read-time metadata writes; a commit per read would be absurd. | -## ❌ Cannot map — always returns `error-code::unsupported` +A hard link to a *directory* returns `not-permitted`, as on Linux: it would make the namespace a graph rather than a tree, and the dnode's parent link has room for exactly one answer. -| WASI op | Why | -|---|---| -| `link-at` (hardlinks) | POSIX requires shared mutability across link members; S3 has no shared-identity-across-keys. The implementable approaches (indirection layer with refcount sidecar, or fan-out copy on every write) all impose either an extra GET per read, O(N) PUTs per write, or fragile crash-recovery state. **Permanent gap by design.** | +## ❌ Not implemented -## Test coverage +Nothing in `wasi:filesystem@0.2.6` is unsupported. + +## Snapshots + +Every root record is a complete, self-verifying snapshot. Copy-on-write means the blocks an older root names were never overwritten, so taking one costs nothing and keeping one costs only the storage its blocks already occupy. + +```rust +let snaps = fs.snapshots(10).await?; // newest first +let snap = fs.open_snapshot(snaps[3].seq).await?; +let ino = snap.lookup(&snap.root(), "deleted-file").await?; +let bytes = snap.read(&ino, 0, 4096).await?; +``` -- **171 unit tests** in `s3fs-core` and `s3fs-wasmtime` against the in-memory backend. -- **10 integration tests** in `s3fs-core/tests/minio_integration.rs` against MinIO via testcontainers (run with `--features aws --test minio_integration -- --ignored`). -- **End-to-end SQLite test**: `examples/guest-fsdemo` runs a real SQLite database on top of the S3-backed filesystem (CREATE TABLE, INSERT, COMMIT, re-open, SELECT, verify) and prints `OK`. +Reads through a snapshot verify exactly as the live mount does — same checksums, same position binding — because a snapshot is not a copy of anything. It is the same blocks, reached through an older root. -## Design references +Opening one does not lower the live mount's rollback floor. Reading history must never become a way to make the store rewind. + +## Operational notes + +- **The bucket is not browsable.** Contents are opaque, encrypted, content-independent blocks. `aws s3 ls` shows slabs and roots, nothing resembling a path. +- **No garbage collection yet.** Copy-on-write means superseded blocks accumulate. Roots are ~700 bytes each and locked for the full retention term; slabs are the cost driver and live in an unlocked bucket precisely so they can be reclaimed later. +- **Keys are supplied, not discovered.** The master secret and the filesystem id are inputs. The keys that verify a root record derive from them, so reading either out of the store would mean trusting the store to say which key checks its own signature. + +## Running SQLite + +SQLite works, including `VACUUM`, triggers, foreign-key cascades, recursive +CTEs, window functions, and `PRAGMA integrity_check`. Two settings are needed, +and neither is a limitation of this filesystem: + +```rust +conn.pragma_update(None, "locking_mode", "EXCLUSIVE")?; // no fcntl under WASI +conn.pragma_update(None, "temp_store", "MEMORY")?; // no access(2) under WASI +``` + +`temp_store=MEMORY` is the non-obvious one. SQLite locates a directory for +temporary databases by probing candidates with `access(2)`, which WASI does not +provide — so every candidate is rejected, and `VACUUM` fails with a bare +`disk I/O error` that says nothing about a missing syscall. Setting +`temp_store_directory` does not help: that pragma validates the path the same +way and reports `not a writable directory`. + +`journal_mode=DELETE` (the default) is fine and worth keeping: the rollback +journal is a real sidecar file, created, extended, truncated and unlinked on +every transaction. WAL is not usable — it needs shared memory that WASI has no +way to provide. + +WAL is not usable — it coordinates through a shared-memory index that WASI has +no way to provide — and concurrent connections are out for the same reason +`locking_mode=EXCLUSIVE` is needed. Neither restricts anything real here: the +store is single-writer by design. + +JSON, FTS5 and R-Tree are all present in the bundled build and all verified +working. [`examples/guest-sqlite`](../examples/guest-sqlite/) covers DDL, +transactions, savepoints, every constraint kind, joins, CTEs, window functions, +blobs including incremental `sqlite3_blob_open` I/O, `ATTACH`, `WITHOUT ROWID`, +generated columns, partial and expression indexes, `DROP`, `VACUUM`, and +`integrity_check` before and after a reopen. The README has the full notes and +the benchmark table. + +## Test coverage -- The race-three lookup, MPU `copyUnmodifiedParts` strategy, and tiered part schedule are direct ports of [GeeseFS](https://github.com/yandex-cloud/geesefs)'s patterns. See [its `core/` directory](https://github.com/yandex-cloud/geesefs/tree/master/core) for the original Go implementation. -- The `wasi:filesystem` WIT lives at [`wit/deps/filesystem.wit`](../wit/deps/filesystem.wit) (vendored from `wasmtime-wasi 44`'s `wasi@0.2.6`). +333 unit tests plus a MinIO integration suite covering the real S3 wire protocol, Object Lock retention, end-to-end write/read/remount, slab packing economics, tamper detection, and mount-floor enforcement. Design lineage: ZFS for the storage model, [GeeseFS](https://github.com/yandex-cloud/geesefs) for the original path-mapping engine this replaced. diff --git a/docs/DEV_ENCLAVE.md b/docs/DEV_ENCLAVE.md new file mode 100644 index 0000000..ab3cda6 --- /dev/null +++ b/docs/DEV_ENCLAVE.md @@ -0,0 +1,228 @@ +# The development enclave + +A real enclave on your own machine, to develop a client against. + +```bash +sudo modprobe vsock_loopback # once per boot +docker build -t s3fs-qemu-nitro:latest deploy/qemu-nitro +cargo install vhost-device-vsock --root target/qemu-nitro/tools + +deploy/qemu-nitro/dev-enclave.sh --guest path/to/your-component.wasm +``` + +It prints the URL and the three values a client must pin, then stays up until +you stop it. + +This is the same stack [`run-e2e.sh`](../deploy/qemu-nitro/run-e2e.sh) asserts +against — both scripts share +[`deploy/qemu-nitro/lib.sh`](../deploy/qemu-nitro/lib.sh), so the thing you +develop against is the thing CI checks. The difference is only what happens +after the enclave starts serving: the e2e asserts and exits, this waits. + +## What is real + +The enclave boots as an EIF under QEMU's `nitro-enclave` machine — the same +image format and the same boot path as production, measured into PCR0 by the +hypervisor. It takes its address by DHCP over emulated vsock, mounts the +encrypted block store over that link, fetches your component from the store and +measures it into PCR16 before it can obtain a key, orders a certificate from a +CA over RFC 8555 TLS-ALPN-01, and gates every request on a WebAuthn assertion +bound to that exact request. + +Your client verifies the attestation document by signature, certificate chain, +validity window, pinned root and both measurements — the same code path, with +the same flags, that it will run against hardware. + +## What is not real + +**The key that signs attestation documents.** QEMU's NSM implements the protocol +but not the signing; its source says *"we don't actually sign the data, so we use +-1 as the 'alg' value"*, and -1 is not a COSE algorithm identifier. So the +emulator image mints a certificate chain at boot and re-signs what the device +produced, contents untouched +([`runtime/src/testing/cosign.rs`](../runtime/src/testing/cosign.rs)). + +The key lives inside an image its operator controls, so a verified document here +means *"this image said so"*, where on hardware it means *"a Nitro enclave with +this measurement said so"*. That gap needs hardware and nothing local can close +it. The root changes at every boot for exactly that reason: there is deliberately +nothing here to hardcode into an app. + +Before this, every client against the emulator had to pass `--unsigned-emulator`, +which skips the signature, the chain and the validity windows — most of what a +client does, and the part most worth exercising before it meets hardware. + +**KMS.** It will not release a key against a document it cannot trace to a Nitro +root, so the emulator image uses a static master key. On hardware, a key policy +pinning PCR0 and PCR16 is what stops a substituted guest from reading anything; +locally, a substituted guest boots and serves. The e2e's leg 8 shows the half +that *is* testable: the enclave measures whatever it fetched, and a client +pinning the approved guest refuses it. + +**The CA.** Certificates come from [Pebble](https://github.com/letsencrypt/pebble), +Let's Encrypt's test server. The protocol is real RFC 8555 — directory, account, +order, TLS-ALPN-01 challenge, finalize, and a certificate sealed into the store — +but no public root signs it, so a client needs Pebble's root as well, printed as +`--ca` on startup. A real client should pin the certificate out of the +attestation document instead, which is what the document binds it for. + +## Must PCR0 always be zero? + +No, and it is not. Under QEMU, PCR0 is a genuine measurement of the image, and +the e2e checks that the value the enclave attests is byte-for-byte the value +`nix build` printed: + +``` +PCR0 build: 8f2026d5a6c50479e27152c06ca86852ce0ec9528efd5f… + attested: 8f2026d5a6c50479e27152c06ca86852ce0ec9528efd5f… +``` + +Change anything in the image — code, or a setting in `eif-qemu`'s environment — +and PCR0 changes, because an enclave image is its configuration as much as its +code. The emulator's PCR0 differs from production's for that reason, which is +the honest outcome: a client can tell which it is talking to. + +PCR0 is all zeroes only in the debug mode `nitro-cli run-enclave --debug-mode` +produces, which this image does not use. The one constant measurement in this +repo is in the in-process test harness +([`runtime/src/testing/mod.rs`](../runtime/src/testing/mod.rs)), which has no +image to measure. + +## What a client pins + +Three values, and none stands in for another: + +| | what it says | why it alone is not enough | +|---|---|---| +| `--pcr0` | the runtime image, measured by the hypervisor | the image no longer contains the guest | +| `--pcr16` | your component, measured by the runtime before it could get a key | the runtime writes it, so a substituted *runtime* could claim your value | +| `--trust-root` | what the document chains to | says nothing about which enclave | + +Pass `--pcr0` and `--pcr16` together: PCR0 says the runtime that wrote PCR16 is +yours, PCR16 says which application it loaded. + +Do **not** pass `--allow-untrusted-root`. With it, verification accepts whatever +root the document arrived with and reports it as self-signed, so a "pinned" root +pins nothing. Without it the presented root must equal yours — the same +comparison a client makes against AWS's. + +## Trying it + +The repo's own client signs a WebAuthn assertion per request, which is the only +way anything reaches the guest: + +```bash +target/release/passkey-client \ + --url https://127.0.0.1:8443 --state target/qemu-nitro/dev/alice.json \ + --trust-root target/qemu-nitro/dev/trust-root.der \ + --pcr0 --pcr16 \ + get --path /counter +``` + +Registration is open: any passkey may enrol and gets its own tenant. A second +`--state` file is a second tenant, which is how to check that one cannot see the +other's data. + +To verify the attestation without touching the guest, `nitro-attest --url +https://127.0.0.1:8443/auth/` with the same three flags. The document rides on +the `/auth/` exchange, which needs no credential — that is where a client +identifies the enclave before it approves anything. + +## Options + +| flag | | +|---|---| +| `--guest COMPONENT.wasm` | the component to serve. Without one, the example in `examples/guest-http` is built | +| `--port PORT` | host port forwarded to the enclave's `:443`. Default 8443 | +| `--name NAME` | names this run's containers and its directory under `target/qemu-nitro`. It keeps runs apart on disk; only one can be up at a time, because MinIO's port and the enclave's vsock CID are fixed | +| `--rp-id DOMAIN` | the WebAuthn relying party the image is built with. Default `enclave.test`; a phone app needs a domain whose `assetlinks.json` names it | +| `--allowed-origin ORIGIN` | an origin assertions may claim besides `https://`, repeatable — `android:apk-key-hash:` for an Android app | +| `--keep-store` | keep the store in `target/qemu-nitro/-store` and resume it on the next start with that name — see [Restarting](#restarting) | +| `--fresh` | with `--keep-store`, discard the kept store first | +| `--guest-egress ORIGIN` | an origin the guest may send requests to, `http(s)://host[:port]`, repeatable. None by default. The machine running the script is `192.168.127.254` from inside the enclave | +| `--background-timeout SECS` | how long one background task may run; the runtime's default is 30 | +| `--guest-env NAME=VALUE` | a variable for the guest, repeatable — e.g. the address of the origin it may reach | + +Everything the run produced lives under `target/qemu-nitro//`: the console +log, the trust root, the passkey state files, and `fcm-messages.jsonl` — every +notification the guest raised, since Firebase is stubbed locally. + +## On a public host + +The same stack runs on a machine the internet can reach, which is how a client is tested against +something other than a laptop — a phone on a mobile network, a staging deployment. **It is test +infrastructure, not a trust boundary.** Everything in [What is not real](#what-is-not-real) still +holds and matters more once the host is not yours alone: whoever controls the host can read every +tenant's data (the master key is static and public in `flake.nix`) and sign any attestation document +(the chain is minted inside the image). Put test money behind it, never real money. + +What changes: + +| flag | | +|---|---| +| `--domain NAME` | serve `NAME` with a certificate from Let's Encrypt, validated over TLS-ALPN-01. `NAME` must resolve to the host and `--port` must be 443, reachable from the internet. Pebble is not started | +| `--acme-staging` | Let's Encrypt's staging CA. Prove issuance with it first: production allows five duplicate certificates a week, and nothing trusts a staging one | +| `--acme-contact EMAIL` | the contact registered with the CA | +| `--fcm-project ID --fcm-service-account FILE` | real Firebase notifications instead of the stub. The key is baked into the image, so it lands in the builder's Nix store and the EIF | +| `--memory SIZE` | the enclave's memory (QEMU `-m`). Default `3G`; the runtime with a small guest runs in `1536M` | +| `--store-bind ADDR` | publish MinIO on one address, e.g. `127.0.0.1`. Its credentials are the defaults; the enclave still reaches it through gvproxy | +| `--publish-hook CMD` | run `CMD ` once the enclave is up. The trust root is new every boot, so whatever clients read their pins from needs the new one | +| `--pack DIR` / `--prebuilt DIR` | build here, run there — below | + +The host needs KVM (bare metal, or an instance with nested virtualisation — on EC2 the c8i, m8i and +r8i families), `vsock_loopback`, Docker, `python3`, `jq`, `curl`, `openssl`, and for an unprivileged +user on :443, `sysctl net.ipv4.ip_unprivileged_port_start=443`. + +### Build here, run there + +Building needs Nix, cargo and this checkout; running needs none of them. `--pack DIR` builds the +image with the image options given, plus every host binary — gvproxy (static), `nitro-attest`, +`passkey-client` and `vhost-device-vsock` (libc only) — into `DIR`, records the image options in +`DIR/image.env`, and stops. On the host: + +``` +dev-enclave.sh --prebuilt DIR --guest component.wasm --port 443 --keep-store ... +``` + +needs only `deploy/qemu-nitro/` and `scripts/minio-up.sh` from this repository beside it, and the +QEMU image loaded (`QEMU_IMAGE` names it). Image options are refused with `--prebuilt`: the image is +what was packed. The build image carries QEMU's compilers; copying `/usr/local` into a slim Debian +image is about 100 MB compressed, against 3.7 GB. + +The attestation chain the emulator mints is valid for 30 days, so a long-running host has to be +restarted within that — with `--keep-store`, a restart costs nothing but new pins. + +## Restarting + +Stop it and start it again with the new component. By default that is a fresh +start, not a reload: the store is rebuilt, so the enclave boots into genesis +rather than resuming, and PCR16 changes. A client still pinning the old +measurement will refuse the new enclave, which is the behaviour you want to see. + +With `--keep-store` the store survives. MinIO's data lives in +`target/qemu-nitro/-store/minio` instead of inside its container, so the +next start with the same name finds the receipt and the store together and +**resumes**: every tenant, passkey and file a guest wrote is still there. A new +component is an **upgrade** of that store — the boot machine records the new +pair — not a new one. Two things still change every boot, and clients have to +re-read them: the trust root, minted fresh by design, and PCR16 whenever the +component did. + +Pebble also starts a new CA every time, while a kept store holds the certificate +an earlier Pebble issued and serves it until renewal. So with `--keep-store` +every root and intermediate seen is kept in the store directory, and +`pebble-root.pem` holds all of them — trust that whole file. The store directory +also carries an `id`, stable across restarts, for clients that keep state per +store. `--fresh` (with `--keep-store`) deletes the directory and starts over; +so does deleting it by hand while the enclave is down. + +## If it does not start + +| | | +|---|---| +| `no /dev/vsock` | `sudo modprobe vsock_loopback` | +| `no /dev/kvm` | the `nitro-enclave` machine needs KVM; it will not run in a VM without nested virtualisation | +| `missing QEMU image` | `docker build -t s3fs-qemu-nitro:latest deploy/qemu-nitro` | +| `... is not tracked by git` | Nix flakes copy only tracked files; `git add deploy/qemu-nitro/` | +| the enclave never reported a trust root | `S3FS_COSIGN_ATTESTATIONS` is unset in the image — check `eif-qemu`'s environment in `flake.nix` | +| a port is in use | `--port`, and `--name` if you want two at once | diff --git a/docs/NOTIFICATIONS.md b/docs/NOTIFICATIONS.md new file mode 100644 index 0000000..23f2eaf --- /dev/null +++ b/docs/NOTIFICATIONS.md @@ -0,0 +1,161 @@ +# Notifications + +A guest can ask the runtime to wake its tenant's devices through Firebase Cloud +Messaging. It is for the moments that matter — work that finished, an approval +somebody is waiting on — and it exists because a guest has no other way to reach +a person between requests. + +The guest does not send anything. It cannot: [`serve::EgressPolicy`] refuses +every outgoing request a guest makes, deliberately. The runtime holds the +credential and owns the connection. + +## A wake signal carries nothing + +**There is no title, no body, and no `notification` object. Not behind a flag.** + +An FCM payload travels through the parent instance — the party an enclave exists +to exclude — and then through Google. Anything put in it is disclosed to both. So +a wake carries an opaque `category`, an optional tenant-local `reference`, and a +schema version: + +```json +{"message": {"token": "…", + "data": {"v": "1", "category": "approval-needed", "ref": "txn-9f2"}, + "android": {"priority": "high", "ttl": "3600s"}, + "apns": {"headers": {"apns-push-type": "background", "apns-priority": "5"}, + "payload": {"aps": {"content-available": 1}}}}} +``` + +The app wakes and fetches the detail over its own attested connection, where the +parent is excluded again. That costs a round trip and is the entire point. + +It is also the right shape mechanically: a `notification` block is rendered by +the OS *without* the app running, so a wake carrying one would show text and fail +to wake anything. + +Labels are identifiers, not prose — 1–64 characters of `[A-Za-z0-9_-]`. If a +category name would itself be sensitive, choose an opaque one; Google sees the +label, the timing and the destination regardless. + +## Enable + +Set a project and exactly one credential source. Empty means off. + +| Setting | Purpose | +|---|---| +| `S3FS_FCM_PROJECT_ID` | The Firebase project. Baked into the image, so PCR0 covers it | +| `S3FS_FCM_SERVICE_ACCOUNT` | The service-account JSON itself. Development | +| `S3FS_FCM_SERVICE_ACCOUNT_PARAMETER` | An SSM parameter holding that JSON. Production | +| `S3FS_FCM_ENDPOINT` | Send somewhere else. Tests and the emulator only | + +Both credential sources at once is refused rather than ranked: a deployment that +set both has one of them wrong, and guessing which is the wrong kind of help. A +literal credential is parsed — key included — at startup, so a malformed one +fails at boot rather than the first time somebody is waiting to be woken. + +Notifications require authentication and tenant isolation, for the same reason +background tasks do: the tenant a wake belongs to comes from a verified +assertion, and without a gate there is none. + +In the Nix deployment set `fcmProjectId` and `fcmServiceAccountParameter` in +`deploy/nix/deployment.nix`. Both change PCR0. + +## Guest contract + +```text +register-device(token) -> result<_, string> interactive only +forget-device(token) -> result<_, string> interactive only +devices() -> result +wake(category, reference) -> result<_, string> background allowed +``` + +No function takes a tenant id. The runtime supplies it from the executing +instance, so a guest cannot name another tenant's devices. + +**Enrolling is interactive-only; waking is not.** That asymmetry is deliberate +and is the one thing here that differs from `enclave:tasks/queue`, where every +mutation is interactive-only. Enrolling a device grants standing ability to reach +somebody, and background work must not be able to grant itself that — nor to +silence its owner by un-enrolling. But raising a wake *from* background work is +the primary use: a task that finishes at three in the morning telling its owner +to come and look. It grants nothing; it spends an enrolment an interactive call +already made. + +`devices()` returns a count and never the tokens. A registration token is a +capability to wake that device from anywhere, so it does not re-enter guest +memory once enrolled. + +## Delivery + +Best effort, and unacknowledged. `wake` returns as soon as the signal is queued; +it never waits on the network, because it runs inside a host call holding the +tenant's only instance slot. + +- Repeat wakes for one `(tenant, category)` **coalesce** — a second signal for a + category still waiting *is* the one already waiting. +- A full queue **drops and counts** rather than blocking. A drop is not something + the guest can act on, so it is not reported to it; exceeding the per-tenant cap + of distinct in-flight categories *is* returned as an error, because that one is + actionable. +- Transient failures retry with backoff. Credential failures count as transient: + an expired token heals on the next refresh, and treating it as fatal would turn + routine rotation into an outage. +- A token FCM reports as `UNREGISTERED` is **pruned, never retried**. It is + terminal by definition, it would spend the tenant's backoff budget while their + other devices queue behind it, and it would occupy one of eight slots for ever. +- Queued wakes are **not durable**. A restart loses them, on purpose: in `tasks` + the record *is* the work, whereas here it would be a stale pointer to work that + already happened, and the next wake or the next poll supersedes it. + +A tenant may enrol eight devices. At the cap the oldest is evicted rather than +the newest refused — this is a cache of places a person can be reached, not a +credential list, and somebody on their ninth phone must still be able to enrol +it. + +## What a stolen credential buys + +The service account reaches the runtime through the parent instance, which is the +party the enclave excludes. State it plainly. + +A parent that steals it **can** send wake signals to registration tokens it +obtains elsewhere, as this project; and it can delay, drop or reorder the +enclave's own sends, which it could already do because it carries every packet. + +It **cannot** read any tenant's data, or the device tokens themselves — those +live in `/runtime/devices` inside the encrypted filesystem, under a key KMS +releases only against a matching PCR0 and PCR16. It cannot impersonate the +enclave to a client, which needs an attestation document it cannot produce. And a +forged wake means nothing, because a wake carries no state: the app's response to +one is to fetch over the attested channel, where the parent is shut out again. + +The credential protects a doorbell. Keeping it in SSM rather than the image means +it can rotate without moving PCR0, which is proportionate to that. + +## The example guest + +`examples/guest-http` exposes: + +| Request | Behavior | +|---|---| +| `POST /devices` | Enrol the body as a registration token | +| `GET /devices` | How many devices this tenant has enrolled | +| `DELETE /devices` | Forget the token in the body | + +and raises `wake("task-done", )` at the end of `run-task`, which is +where the interactive/background split is visible in practice. + +## Build and test + +```sh +cargo build --manifest-path examples/guest-http/Cargo.toml --release --target wasm32-wasip2 +cargo test -p enclave-runtime --lib notify:: +cargo test -p enclave-runtime --lib tasks::tests -- --include-ignored +``` + +Nothing in the suite reaches Google: the wire is covered by a transport double +that records exactly what would have been sent. **The FCM path is unverified +against the real service** — the same honesty `guest_io::cloudwatch` applies to +its own credential path. What is exercised end to end is leg `6b/8` of +[`deploy/qemu-nitro/run-e2e.sh`](../deploy/qemu-nitro/run-e2e.sh), where a +finished background task wakes an enrolled device inside an emulated enclave and +a stub records that the message carried no content. diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md new file mode 100644 index 0000000..3b8403b --- /dev/null +++ b/docs/ROADMAP.md @@ -0,0 +1,229 @@ +# Roadmap + +The storage engine is complete (M0–M7, `b11f95b`) and so is the runtime that +loads a guest and gives it the filesystem (M10). What remains is the enclave's +own plumbing — proving to KMS which code is running — and reclaiming space. + +| | | Status | +|---|---|---| +| M0–M7 | Merkle-anchored block store, POSIX layer, hard links, snapshots | ✅ done | +| **M8** | [NSM attestation and KMS key release](#m8--attestation-and-key-release) | needs enclave hardware | +| **M9** | [Garbage collection](#m9--garbage-collection) | not started | +| M10 | [`enclave-runtime`](#m10--the-enclave-runtime-) | ✅ done | + +--- + +## M8 — Attestation and key release + +Today the master secret arrives via `--master-key`. That is a development +seam and it is the last thing standing between this and a real enclave +deployment: a secret passed on the command line is visible to the parent +instance, which is precisely the party the enclave exists to exclude. + +The on-disk format does not change. `KeyMaterial::derive` takes 32 bytes and +does not care where they came from, which was the point of shaping it that way +in M1. + +### What has to be built + +**1. NSM attestation documents — done.** [`nitro-nsm`](../crates/nitro-nsm/src/lib.rs) +issues the `Attestation` request and [`nitro-attestation`](../crates/nitro-attestation/src/lib.rs) +parses and verifies the COSE_Sign1 that comes back, against the AWS Nitro root. +The runtime already binds its TLS certificate and the guest into `user_data` +and puts a document on every `/auth/*` response, in `x-enclave-attestation`. + +What M8 still needs from this layer is the `public_key` field, which is +currently always absent: KMS `Decrypt` with `Recipient` returns the plaintext +encrypted to a key the enclave generates at boot, and that key goes there. + +**2. KMS `Decrypt` with `Recipient`.** The attestation document goes in the +`Recipient` field; KMS returns the plaintext encrypted to the enclave's public +key rather than in the clear, so the parent proxying the call never sees it. +The key policy binds release to the PCRs: + +```json +{ + "Effect": "Allow", + "Action": ["kms:GenerateDataKey", "kms:Decrypt"], + "Condition": { + "StringEqualsIgnoreCase": { + "kms:RecipientAttestation:PCR0": "", + "kms:RecipientAttestation:PCR16": "" + } + } +} +``` + +PCR0 covers the runtime image, and the guest is not in it. The runtime fetches +the guest at boot, extends PCR16 with its hash and locks the register before it +asks for the key, so the key is still released only to exactly this code — +runtime *and* guest — while a guest change is a policy edit rather than an +image rebuild. Neither condition is enough alone: PCR0 alone admits any guest, +and PCR16 alone admits any runtime that writes the right value into the +register. + +**3. A vsock HTTP client.** An enclave has no network except vsock, so both +KMS and S3 go through a proxy on the parent. This needs a seam that does not +exist: + +- [`AwsS3BackendConfig`](../crates/s3fs-core/src/backend/aws.rs) has no + `http_client` field, so the SDK cannot be pointed at a vsock connector. + Add `http_client: Option` and thread it into + `connect_unchecked`. +- Static credentials there are built with `None` expiry + ([aws.rs](../crates/s3fs-core/src/backend/aws.rs)), so a KMS-attested + session token expires mid-run with no recovery. Needs a refresh path. + +**3c. Prove the enclave can reach IMDS.** The runtime's AWS clients resolve +credentials through the SDK's default chain, which reaches `169.254.169.254` on +the parent by way of gvproxy. Nothing exercises that: the QEMU harness has no +metadata service, so the path is assumed rather than tested, and S3, KMS, SSM +and CloudWatch all rest on it. Validate IMDSv2 resolution on Nitro hardware, and +settle `http_put_response_hop_limit` (`deploy/tofu/main.tf`) — 1 if gvproxy +proxies, 2 if it routes. The failure mode is misleading: calls fail as though +credentials were missing rather than as though the network were broken. + +**3b. Cover the clock.** The guest's wall clock now comes from `/dev/ptp0` +(see the README), but two things remain. The Object Lock retention deadline in +[`RootStore::root_retention`](../crates/s3fs-core/src/store/root.rs) still uses +`SystemTime::now()` — and a COMPLIANCE deadline computed from a wrong clock +cannot be corrected afterwards by anyone, which makes it the sharpest version +of this problem in the codebase. And attestation should eventually cover *which* +clock was in use, so a relying party can tell a PTP-backed enclave from one that +fell back to host time. + +**3c. Read past delete markers — done.** Object Lock protects a *version*, not +the name it lives under: a `DeleteObject` with no version id writes a delete +marker, and `HEAD`/`GetObject` then report an Object-Locked record missing while +it sits underneath, undeletable. That made "hide the tip" a legal S3 call rather +than a lie from S3, and made a hidden state-origin receipt indistinguishable from +none — which authorises genesis. [`Backend::get_retained_blob`](../crates/s3fs-core/src/backend/mod.rs) +reads the version through `ListObjectVersions`; the boot machine and +`RootStore::exists`/`load` use it. Needs `s3:ListBucketVersions` on the enclave's +role, and errors rather than answering "absent" without it. + +**4. Bind the attestation into the anchor.** *Partly done, differently.* +`enclave_runtime::boot` writes a state-origin receipt at genesis — an NSM +attestation committing to the filesystem's identity — and every later boot +verifies it, so *which code created this state* is now established. `RootRecord` +has no `attestation` field (an earlier version of this document claimed it did); +recording a PCR digest per *commit* would turn the chain into a full audit log +rather than a statement about the origin, and would put an NSM signature in the +write path of every transaction. + +**5. Close the cold-mount gap.** The *substitution* half is closed: a store +with no filesystem no longer yields a new one, because genesis needs a receipt +to be absent and Object Lock keeps it present. The *rollback* half remains — +`--min-root-seq` is still supplied by hand. Carrying it in the KMS encryption context would make the floor something +the enclave receives from a service the storage operator does not control — +the one thing that closes the residual risk documented in +[`store/root.rs`](../crates/s3fs-core/src/store/root.rs). + +### Testing + +Everything except the NSM calls can be tested without hardware: the vsock +client against a local listener, the credential refresh against MinIO, and the +key-source abstraction with a fake that returns fixed bytes. + +The NSM calls themselves now have a home too. [`deploy/qemu-nitro/`](../deploy/qemu-nitro/) +builds an EIF and boots it on QEMU's `nitro-enclave` machine, where +`GetRandom` against the emulated device already passes. Attestation should +extend that harness rather than wait for hardware: `eif_build` produces genuine +PCR0/1/2, so the measurements an `Attestation` response must quote are already +known values. What the emulator cannot give is a document signed by the real +AWS Nitro root, so certificate-chain validation and the KMS `Recipient` round +trip still need `nitro-cli run-enclave`, and CI cannot cover either. + +--- + +## M9 — Garbage collection + +Copy-on-write means every superseded block is still there. Nothing reclaims +them, and nothing can until there is a way to decide what is dead. + +The split buckets already put the cost where it can be addressed: root records +are ~700 bytes and locked for the full retention term, which is cheap forever; +slabs are the volume and live unlocked precisely so they *can* be deleted. + +### The hard part + +A slab is dead when no root you intend to keep references any block in it. +"Intend to keep" is a policy decision, not a fact about the data, and getting +it wrong deletes live data. Hence: not shipped rather than shipped +approximately. + +**It needs a home first.** `s3fs-runner` was deleted once `enclave-runtime` +became a superset of it, so there is currently no binary for operator tooling — +and GC should not go in `enclave-runtime`. That binary ships *inside* the +enclave image, so every flag added to it changes PCR0 and enlarges the attested +surface. Sweeping dead blocks is something an operator does to a store from +outside; it has no business being measured as part of the code a client +attests. M9 starts by adding a small unattested CLI for this and anything like +it (`fsck`, `show-root`, `verify-chain`). + +Proposed `gc --keep-roots N`: + +1. Mark: walk the blkptr trees of the newest `N` roots, collecting live + `(txg, slab)` pairs. Bounded by live data, not by history. +2. Sweep: `ListObjectsV2` over `slabs/`, delete any slab in no live set. +3. Never touch the roots bucket. It stays a complete, immutable, ten-year + audit log of state hashes regardless — GC trades away the ability to + *replay* old states, not the record that they existed. + +Two things fold in naturally: + +- **Orphaned dnodes.** A crash while a file is unlinked-but-open leaves an + allocated, nameless, unreachable dnode. The mark phase already knows what is + reachable from the root directory, so these fall out for free. +- **Orphaned slabs.** Slabs from commits that died before publishing a root. + They are unreferenced by construction, and their transaction group is + recorded by the next session's claim root, so nothing ever writes there. + +### Cheaper stopgaps + +- Raise `record_size` for write-heavy workloads: fewer, larger blocks means + less metadata churn per byte rewritten. +- A lifecycle rule expiring `slabs/` objects older than the oldest root you + care to keep. Crude and unsafe in general — a rarely-touched block can be + both old and live — but usable if the workload rewrites everything + periodically. + +--- + +## M10 — The enclave runtime ✅ + +Shipped. `enclave-runtime` mounts the store, loads a guest from a known path, +and runs it. The library half of that crate holds the `wasi:filesystem` implementation, the linker, +the guest-environment policy, and the run loop. + +``` +crates/ + s3fs-core engine + nitro-nsm /dev/nsm: entropy and attestation requests + nitro-attestation parse and verify documents; the nitro-attest client + enclave-runtime deployment target +``` + +Configuration is environment-first with matching flags, because inside an +enclave the image's `ENV` lines are the only configuration there is. The guest +inherits that environment minus anything under `AWS_` or `S3FS_` — credentials +and the runtime's own settings — which is the one rule that keeps the master +key away from the code the enclave exists to contain. + +`deploy/Dockerfile` builds an image without the guest. The runtime fetches it +from the roots bucket at boot, measures it into PCR16 and locks the register +before it asks for a key, so M8's key policy — PCR0 and PCR16 — attests to the +code that reads the data rather than only to the runtime that loads it. + +### What M8 changes here + +Almost nothing structural, by design: + +- `StaticKey` becomes `KmsAttestedKey` — one more implementation of + `enclave_runtime::MasterKeySource`, and `S3FS_MASTER_KEY` stops being read at all. +- `AwsS3BackendConfig` gains the `http_client` field so the SDK can be pointed + at the parent's vsock proxy, plus a credential refresh path. +- `RootRecord::attestation` starts carrying the PCR digest, turning the anchor + chain into a record of *which code* wrote each state. +- `S3FS_MIN_ROOT_SEQ` moves into the KMS encryption context, so the freshness + floor comes from a service the storage operator does not control. diff --git a/docs/STREAMING.md b/docs/STREAMING.md new file mode 100644 index 0000000..d88ef76 --- /dev/null +++ b/docs/STREAMING.md @@ -0,0 +1,157 @@ +# Bidirectional streams + +A signing session is interactive: each round's messages depend on the last +round's replies. That needs a channel open in both directions at once, not a +sequence of requests. This is how one works here, what opening one is allowed +to authorize, and what it costs. + +The prototype carries **no cryptography**. `examples/guest-grpc` exchanges +dummy signing-session messages so the transport and the authorization are +settled before a key depends on either. + +## What works + +A guest holds the request body and the response body at the same time and +answers each message as it arrives. This is specified behaviour rather than a +trick: `wasi:http`'s `response-outparam.set` is documented to *"allow execution +to continue after the response has been sent"*, and `incoming-request.consume` +borrows the request rather than consuming it, so the two resource trees are +independent. + +The runtime negotiates HTTP/2 by ALPN, per connection. There is no +configuration for it, and HTTP/1.1 clients are unaffected. + +## Authorization: one model, and what it names + +A stream and an ordinary request are approved the same way. That unification is +recent and deliberate: a stream's body **is** its message sequence, so it could +never be bound by a body hash, and rather than keep a second weaker path for it, +the approval now names something both shapes have — the interaction. + +```console +POST /auth/request/options { credential_id, method, path, query } + ↓ response is attested: check it, pin the certificate + Face ID / Touch ID + ↓ +POST /auth/request/verify { challenge_id, assertion } → { token, expires_in_secs } + ↓ + Authorization: Bearer +``` + +### What it authorizes + +**One interaction, at that method, path and query.** Not one message, and not +one payload. + +The token commits to no bytes. A token issued for `POST /sign` authorizes +whatever body follows, so a client compromised between the approval and the +request can substitute the payload — the runtime will not catch it, because it +is no longer looking. That is the trade this model makes, and it is stated here +rather than left to be discovered. + +What the runtime still refuses: moving a token to another route, spending one +twice, spending one after it expires, and spending one belonging to another +tenant. Redeeming removes it before anything about it is checked, so two callers +racing one token have exactly one winner and a token offered for the wrong route +is spent by the attempt. + +### Per-message approval: not built, and not buildable today + +If a guest wants "the person approved *these bytes*", it must obtain that +itself, per message, inside the interaction. **The runtime cannot do it**: it +hands the body through without reading it, and teaching it to parse a guest's +message format would make it care about a protocol it deliberately knows +nothing about. + +**But the guest cannot do it either, yet.** Verifying a passkey assertion needs +the stored credential, the challenge that was issued, the relying-party +configuration, and the ability to mark that challenge used. All four live in the +runtime, and a guest is given standard WASI and nothing else — there is no host +function it can call to ask. + +`examples/guest-grpc` deliberately carries **no** approval fields on +`ClientMsg`. An earlier draft carried `challenge_id` and `assertion` and refused +messages without them, which promised a check nothing performed; fields that +look like a credential and are never verified read as a guarantee, and are worse +than their absence. + +Closing this needs a new host import — a way for a guest to hand the runtime an +assertion and get back yes or no. **It should be designed before a real key +depends on it**, because until then "someone opened an interaction" is the only +thing between a compromised client and whatever the guest will do. + +### What a stolen token buys + +It is single-use, expires in `--interaction-token-ttl-secs` (60s by default), and +is bound to one route. Spending one gets: one interaction with that tenant's +guest, and whatever the guest will do +without a further check. It gets nothing about another tenant, and no replay — +the token is gone the moment it is used. + +## What it costs: one stream occupies one tenant slot + +Concurrency is **one active guest handler per tenant**. That is the isolation +model, not a limit to be raised — a tenant's SQLite database is only safe +because exactly one of their requests is ever in flight (WASI has no `fcntl`, +so `locking_mode=EXCLUSIVE` is the only option). + +A stream is a request. **An open stream holds its tenant's slot for its entire +life**, and that tenant's next request waits behind it. Different tenants are +unaffected and run concurrently. + +Anonymous callers share a single lock, so a stream with no gate configured +would starve every other anonymous caller. **Streams are refused outright when +no gate is configured.** + +## Lifetime and failure + +| event | what happens | +|---|---| +| **half-close** (client sends END_STREAM) | the orderly end: the guest finishes and emits `grpc-status` trailers | +| **client cancels or disconnects** | the guest's next write fails, the call ends, the instance is dropped and never reused, the tenant lock frees | +| **head timeout** (`request_timeout`) | bounds only the *response head*. It does not apply once the head is out, so it never ends a healthy stream | +| **no traffic either way** | the epoch watchdog asks whether bytes moved, not how long the call ran. Sustained silence while executing wasm ends the call | +| **interaction deadline** (`--max-interaction-secs`, 300s) | a wall clock, checked whether or not wasm is executing — so it reaches a guest blocked in a host call, which the epoch never can. Ends the interaction and frees the tenant | +| **guest spins** | trapped by the epoch, on the same schedule as before streams existed | +| **guest traps** | the instance is discarded, never reused — a partially-executed call leaves a store that cannot be re-entered | +| **`grpc-timeout` header** | **not enforced by the runtime.** Forwarded to the guest, which owns it. The runtime's deadlines are its own, so a client cannot lengthen them | +| **`wasi:http` 600s between-bytes** | hardcoded in `wasmtime-wasi-http` and not configurable. A ceiling above ours | + +## Sessions, retries, duplicates + +The `session_id` is the client's; the runtime never reads it. A stream that +dies takes its in-memory session with it — **nothing about a session survives a +stream except what the guest committed to storage**, which is the same rule as +every other request. + +The runtime does not retry, resume, or deduplicate anything. **There is no +exactly-once guarantee and no automatic replay of signing rounds.** A client +that reconnects opens a *new* stream with a *new* stream-open assertion, and +any message it re-sends is simply a new message. +Duplicates are the guest's to detect by sequence number, and its round handling +must be idempotent because the transport promises nothing. + +## Attestation + +The `x-enclave-attestation` document is on the response head, which for gRPC is +the initial metadata. A client should verify it — against the nonce it chose +and the leaf from its own handshake — **before sending anything +signing-sensitive**. + +One document per stream, at open. **It attests the connection**: that this +enclave terminates it and is alive now. It says nothing about individual +messages and nothing about signing results, because it is generated before the +guest runs. + +## Known limits + +- **One task, alternating.** The guest reads and writes in turn rather than + truly simultaneously. Sufficient for a request/response signing protocol; a + guest needing to emit unsolicited frames while blocked on a read would need + more. +- **Per-message authorization is designed, not built.** See above. +- **`grpc-timeout` is forwarded, never enforced.** The runtime's deadlines are + its own so a client cannot lengthen them by asking; interpreting a client's + deadline is the guest's, because only the guest knows what its work is worth. +- **HTTP/1.1 trailers need a `Trailer:` header** or hyper drops them silently. + The guest sets one. HTTP/2 needs no such thing. diff --git a/examples/guest-fsdemo/.gitignore b/examples/guest-fsdemo/.gitignore deleted file mode 100644 index ea8c4bf..0000000 --- a/examples/guest-fsdemo/.gitignore +++ /dev/null @@ -1 +0,0 @@ -/target diff --git a/examples/guest-fsdemo/Cargo.lock b/examples/guest-fsdemo/Cargo.lock deleted file mode 100644 index f2ec5fe..0000000 --- a/examples/guest-fsdemo/Cargo.lock +++ /dev/null @@ -1,196 +0,0 @@ -# This file is automatically @generated by Cargo. -# It is not intended for manual editing. -version = 4 - -[[package]] -name = "ahash" -version = "0.8.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a15f179cd60c4584b8a8c596927aadc462e27f2ca70c04e0071964a73ba7a75" -dependencies = [ - "cfg-if", - "once_cell", - "version_check", - "zerocopy", -] - -[[package]] -name = "bitflags" -version = "2.11.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4512299f36f043ab09a583e57bceb5a5aab7a73db1805848e8fef3c9e8c78b3" - -[[package]] -name = "cc" -version = "1.2.61" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d16d90359e986641506914ba71350897565610e87ce0ad9e6f28569db3dd5c6d" -dependencies = [ - "find-msvc-tools", - "shlex", -] - -[[package]] -name = "cfg-if" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" - -[[package]] -name = "fallible-iterator" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" - -[[package]] -name = "fallible-streaming-iterator" -version = "0.1.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" - -[[package]] -name = "find-msvc-tools" -version = "0.1.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" - -[[package]] -name = "guest-fsdemo" -version = "0.1.0" -dependencies = [ - "rusqlite", -] - -[[package]] -name = "hashbrown" -version = "0.14.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e5274423e17b7c9fc20b6e7e208532f9b19825d82dfd615708b70edd83df41f1" -dependencies = [ - "ahash", -] - -[[package]] -name = "hashlink" -version = "0.9.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ba4ff7128dee98c7dc9794b6a411377e1404dba1c97deb8d1a55297bd25d8af" -dependencies = [ - "hashbrown", -] - -[[package]] -name = "libsqlite3-sys" -version = "0.30.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2e99fb7a497b1e3339bc746195567ed8d3e24945ecd636e3619d20b9de9e9149" -dependencies = [ - "cc", - "pkg-config", - "vcpkg", -] - -[[package]] -name = "once_cell" -version = "1.21.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" - -[[package]] -name = "pkg-config" -version = "0.3.33" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" - -[[package]] -name = "proc-macro2" -version = "1.0.106" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "quote" -version = "1.0.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41f2619966050689382d2b44f664f4bc593e129785a36d6ee376ddf37259b924" -dependencies = [ - "proc-macro2", -] - -[[package]] -name = "rusqlite" -version = "0.32.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7753b721174eb8ff87a9a0e799e2d7bc3749323e773db92e0984debb00019d6e" -dependencies = [ - "bitflags", - "fallible-iterator", - "fallible-streaming-iterator", - "hashlink", - "libsqlite3-sys", - "smallvec", -] - -[[package]] -name = "shlex" -version = "1.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fda2ff0d084019ba4d7c6f371c95d8fd75ce3524c3cb8fb653a3023f6323e64" - -[[package]] -name = "smallvec" -version = "1.15.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67b1b7a3b5fe4f1376887184045fcf45c69e92af734b7aaddc05fb777b6fbd03" - -[[package]] -name = "syn" -version = "2.0.117" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e665b8803e7b1d2a727f4023456bbbbe74da67099c585258af0ad9c5013b9b99" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "unicode-ident" -version = "1.0.24" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" - -[[package]] -name = "vcpkg" -version = "0.2.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" - -[[package]] -name = "version_check" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" - -[[package]] -name = "zerocopy" -version = "0.8.48" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eed437bf9d6692032087e337407a86f04cd8d6a16a37199ed57949d415bd68e9" -dependencies = [ - "zerocopy-derive", -] - -[[package]] -name = "zerocopy-derive" -version = "0.8.48" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "70e3cd084b1788766f53af483dd21f93881ff30d7320490ec3ef7526d203bad4" -dependencies = [ - "proc-macro2", - "quote", - "syn", -] diff --git a/examples/guest-fsdemo/Cargo.toml b/examples/guest-fsdemo/Cargo.toml deleted file mode 100644 index b24c73c..0000000 --- a/examples/guest-fsdemo/Cargo.toml +++ /dev/null @@ -1,22 +0,0 @@ -[package] -name = "guest-fsdemo" -version = "0.1.0" -edition = "2021" -publish = false - -# Standalone — not part of the host workspace, so cargo doesn't try to -# build it for the host platform on `cargo build --workspace`. -[workspace] - -[dependencies] -rusqlite = { version = "0.32", features = ["bundled"] } - -[[bin]] -name = "guest-fsdemo" -path = "src/main.rs" - -[profile.release] -opt-level = "s" -lto = true -codegen-units = 1 -panic = "abort" diff --git a/examples/guest-fsdemo/src/main.rs b/examples/guest-fsdemo/src/main.rs deleted file mode 100644 index c0cc9ca..0000000 --- a/examples/guest-fsdemo/src/main.rs +++ /dev/null @@ -1,141 +0,0 @@ -//! `guest-fsdemo` — Wasm component that exercises `wasi:filesystem` -//! end-to-end, including running a real SQLite database on top of it. -//! -//! Build: -//! ```bash -//! # Requires wasi-sdk for the bundled SQLite C build: -//! # curl -sLO https://github.com/WebAssembly/wasi-sdk/releases/\ -//! # download/wasi-sdk-25/wasi-sdk-25.0-x86_64-linux.tar.gz -//! # mkdir -p ~/wasi-sdk && tar xf wasi-sdk-25.0-x86_64-linux.tar.gz \ -//! # -C ~/wasi-sdk --strip-components=1 -//! CC_wasm32_wasip2=$HOME/wasi-sdk/bin/clang \ -//! AR_wasm32_wasip2=$HOME/wasi-sdk/bin/ar \ -//! CFLAGS_wasm32_wasip2="--sysroot=$HOME/wasi-sdk/share/wasi-sysroot \ -//! -DSQLITE_THREADSAFE=0 -DHAVE_USLEEP=1" \ -//! cargo build --release --target wasm32-wasip2 -//! ``` -//! -//! Run via the `s3fs-runner` against MinIO or AWS. The component prints -//! `OK` on success or `FAIL: ...` on the first failure. - -use std::fs; -use std::io::{Read, Write}; - -fn run() -> std::io::Result<()> { - let root = "/"; - - let dir = format!("{root}data"); - let _ = fs::remove_dir(&dir); - fs::create_dir(&dir)?; - - let path = format!("{dir}/hello.txt"); - let mut f = fs::File::create(&path)?; - f.write_all(b"hello, ")?; - f.write_all(b"wasm world")?; - f.sync_all()?; - drop(f); - - let mut f = fs::File::open(&path)?; - let mut body = String::new(); - f.read_to_string(&mut body)?; - if body != "hello, wasm world" { - return Err(std::io::Error::other(format!( - "read mismatch: {body:?}" - ))); - } - - // Step 4: read_dir - let mut names: Vec = fs::read_dir(&dir)? - .filter_map(|e| e.ok()) - .map(|e| e.file_name().to_string_lossy().into_owned()) - .collect(); - names.sort(); - if names != vec!["hello.txt"] { - return Err(std::io::Error::other(format!( - "read_dir mismatch: {names:?}" - ))); - } - - // Step 5: rename file (file rename is supported; dir rename isn't yet) - let renamed = format!("{dir}/hi.txt"); - let _ = fs::remove_file(&renamed); - fs::rename(&path, &renamed)?; - let mut after_rename = String::new(); - fs::File::open(&renamed)?.read_to_string(&mut after_rename)?; - if after_rename != "hello, wasm world" { - return Err(std::io::Error::other("content lost across rename")); - } - - // Step 6: SQLite. Our wasi:filesystem now hosts a real database — open, - // create a table, insert some rows, query, close. SQLite hammers fsync - // and small reads/writes, so this is a tougher test than std::fs alone. - let db_path = format!("{dir}/demo.db"); - let _ = fs::remove_file(&db_path); - { - let conn = rusqlite::Connection::open(&db_path) - .map_err(|e| std::io::Error::other(format!("sqlite open: {e}")))?; - // We don't implement file locking. Tell SQLite this is the only - // process so it skips lock syscalls. Use in-memory journal to avoid - // sidecar files. `synchronous` is a setter pragma (no return rows), - // so we run it via execute_batch alongside the others. - conn.execute_batch( - "PRAGMA locking_mode = EXCLUSIVE;\ - PRAGMA journal_mode = MEMORY;\ - PRAGMA synchronous = FULL;", - ) - .map_err(|e| std::io::Error::other(format!("sqlite pragmas: {e}")))?; - conn.execute_batch( - "CREATE TABLE notes(id INTEGER PRIMARY KEY, body TEXT NOT NULL); - INSERT INTO notes(body) VALUES ('one'), ('two'), ('three');", - ) - .map_err(|e| std::io::Error::other(format!("sqlite write: {e}")))?; - let count: i64 = conn - .query_row("SELECT COUNT(*) FROM notes", [], |row| row.get(0)) - .map_err(|e| std::io::Error::other(format!("sqlite count: {e}")))?; - if count != 3 { - return Err(std::io::Error::other(format!( - "sqlite count mismatch: {count}" - ))); - } - } - // Re-open and verify the data persisted (i.e. our sync actually flushed - // to S3 and the next open's reads see it). - { - let conn = rusqlite::Connection::open(&db_path) - .map_err(|e| std::io::Error::other(format!("sqlite reopen: {e}")))?; - conn.execute_batch("PRAGMA locking_mode = EXCLUSIVE;") - .map_err(|e| std::io::Error::other(format!("sqlite reopen pragma: {e}")))?; - let bodies: Vec = conn - .prepare("SELECT body FROM notes ORDER BY id") - .and_then(|mut s| { - s.query_map([], |row| row.get::<_, String>(0))? - .collect::>() - }) - .map_err(|e| std::io::Error::other(format!("sqlite read: {e}")))?; - if bodies != vec!["one", "two", "three"] { - return Err(std::io::Error::other(format!( - "sqlite content mismatch: {bodies:?}" - ))); - } - } - - // Step 7: cleanup - fs::remove_file(&db_path)?; - fs::remove_file(&renamed)?; - fs::remove_dir(&dir)?; - - if fs::metadata(&renamed).is_ok() { - return Err(std::io::Error::other("file still exists after delete")); - } - Ok(()) -} - -fn main() { - match run() { - Ok(()) => println!("OK"), - Err(e) => { - eprintln!("FAIL: {e}"); - std::process::exit(1); - } - } -} diff --git a/examples/guest-grpc/Cargo.lock b/examples/guest-grpc/Cargo.lock new file mode 100644 index 0000000..675f7b5 --- /dev/null +++ b/examples/guest-grpc/Cargo.lock @@ -0,0 +1,340 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + +[[package]] +name = "async-task" +version = "4.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b75356056920673b02621b35afd0f7dda9306d03c79a30f5c56c44cf256e3de" + +[[package]] +name = "bitflags" +version = "2.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" + +[[package]] +name = "bytes" +version = "1.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" + +[[package]] +name = "cfg-if" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" + +[[package]] +name = "either" +version = "1.18.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" + +[[package]] +name = "fastrand" +version = "1.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e51093e27b0797c359783294ca4f0a911c270184cb10f85783b118614a1501be" +dependencies = [ + "instant", +] + +[[package]] +name = "futures-core" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" + +[[package]] +name = "futures-io" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" + +[[package]] +name = "futures-lite" +version = "1.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "49a9d51ce47660b1e808d3c990b4709f2f415d928835a17dfd16991515c46bce" +dependencies = [ + "fastrand", + "futures-core", + "futures-io", + "memchr", + "parking", + "pin-project-lite", + "waker-fn", +] + +[[package]] +name = "guest-grpc" +version = "0.1.0" +dependencies = [ + "bytes", + "http-body", + "http-body-util", + "prost", + "wstd", +] + +[[package]] +name = "http" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" +dependencies = [ + "bytes", + "itoa", +] + +[[package]] +name = "http-body" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" +dependencies = [ + "bytes", + "http", +] + +[[package]] +name = "http-body-util" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" +dependencies = [ + "bytes", + "futures-core", + "http", + "http-body", + "pin-project-lite", +] + +[[package]] +name = "instant" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e0242819d153cba4b4b05a5a8f2a7e9bbf97b6055b2a002b395c96b5ff3c0222" +dependencies = [ + "cfg-if", +] + +[[package]] +name = "itertools" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2b192c782037fadd9cfa75548310488aabdbf3d2da73885b31bd0abd03351285" +dependencies = [ + "either", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "parking" +version = "2.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f38d5652c16fde515bb1ecef450ab0f6a219d619a7274976324d5e377f7dceba" + +[[package]] +name = "pin-project-lite" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "prost" +version = "0.14.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "528ac67416ff8646872a3c02cad9cc4ee5dc9f9540c9b10771855c95cb2e5ae1" +dependencies = [ + "bytes", + "prost-derive", +] + +[[package]] +name = "prost-derive" +version = "0.14.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf" +dependencies = [ + "anyhow", + "itertools", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.5", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "slab" +version = "0.4.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" + +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "12df2e0110f65b775f769bb17ef989067a1d931b2eb822bd4346631eeada89f9" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "unicode-ident" +version = "1.0.24" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" + +[[package]] +name = "waker-fn" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "317211a0dc0ceedd78fb2ca9a44aed3d7b9b26f81870d485c07122b4350673b7" + +[[package]] +name = "wasip2" +version = "1.0.4+wasi-0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" +dependencies = [ + "wit-bindgen", +] + +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" +dependencies = [ + "bitflags", +] + +[[package]] +name = "wstd" +version = "0.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29b52936db10a79bb724dadd1c2c3aac958e8229dcb1f1c7f2b7044ca9fc6a3a" +dependencies = [ + "anyhow", + "async-task", + "bytes", + "futures-lite", + "http", + "http-body", + "http-body-util", + "itoa", + "pin-project-lite", + "serde", + "serde_json", + "slab", + "wasip2", + "wstd-macro", +] + +[[package]] +name = "wstd-macro" +version = "0.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "153db9b65508bf6c2efe26617169840c03671ba5719a35868db7682a5261f7d4" +dependencies = [ + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/examples/guest-grpc/Cargo.toml b/examples/guest-grpc/Cargo.toml new file mode 100644 index 0000000..5196c96 --- /dev/null +++ b/examples/guest-grpc/Cargo.toml @@ -0,0 +1,37 @@ +# A workspace of its own, with its own lock file and its own target directory, +# so a guest cannot share the runtime's dependency layer — the same arrangement +# `examples/guest-http` uses and for the same reason. +[workspace] + +[package] +name = "guest-grpc" +version = "0.1.0" +edition = "2021" +publish = false + +[[bin]] +name = "guest-grpc" +path = "src/main.rs" + +[dependencies] +wstd = "0.6" +# Protobuf encoding without a code generator. `prost` is pure Rust and builds +# for wasm32-wasip2; `prost-derive` is a proc macro and runs on the host. There +# is no `protoc` in this build and no build script: the messages are four +# fields each, and the `.proto` beside them is the specification a client +# generates from, not an input to this crate. +# +# `tonic` is absent because tonic does not build for wasm32-wasip2 at all. What +# it would have provided is the gRPC framing, which is five bytes. +prost = { version = "0.14", default-features = false, features = ["derive", "std"] } +bytes = "1" +http-body = "1" +# `UnsyncBoxBody` (the request body, held while the response streams) and the +# combinators for the empty-with-trailers refusal. Same versions wstd resolves, +# so the `http_body::Body` impls are the same trait. +http-body-util = "0.1" + +[profile.release] +opt-level = "s" +lto = true +codegen-units = 1 diff --git a/examples/guest-grpc/proto/enclave/cosign/v1/session.proto b/examples/guest-grpc/proto/enclave/cosign/v1/session.proto new file mode 100644 index 0000000..71dd609 --- /dev/null +++ b/examples/guest-grpc/proto/enclave/cosign/v1/session.proto @@ -0,0 +1,55 @@ +// The wire format this guest speaks. Checked in as the specification a client +// generates stubs from — nothing in this workspace compiles it, because four +// fields of `#[derive(prost::Message)]` is less machinery than a protoc +// toolchain in every build. +// +// The messages are deliberately about a *signing session* without being about +// signing: this prototype settles the transport and the authorization, and +// carries no cryptography at all. +syntax = "proto3"; + +package enclave.cosign.v1; + +service SigningSession { + // One long-lived channel. Both directions are open at once: the guest + // answers each message as it arrives rather than waiting for the client to + // finish, which is the property the runtime exists to support here. + rpc Sign(stream ClientMsg) returns (stream ServerMsg); +} + +enum Kind { + KIND_UNSPECIFIED = 0; + KIND_HELLO = 1; + KIND_ROUND = 2; + KIND_FINISH = 3; +} + +message ClientMsg { + // The client's, never the runtime's. A stream that dies takes its in-memory + // session with it; only what the guest committed to storage survives. + string session_id = 1; + uint64 seq = 2; + Kind kind = 3; + bytes payload = 4; + + // There is deliberately **no per-message approval field here.** + // + // An earlier draft carried `challenge_id` and `assertion`, which promised a + // check nothing performed: verifying a passkey assertion needs the stored + // credential, the issued challenge and the relying-party configuration, all + // of which live in the runtime and none of which a guest can reach. Fields + // that look like a credential and are never checked are worse than no fields + // — they read as a guarantee. + // + // A cosigner that needs "the person approved *these bytes*" must obtain it + // per message, and that needs a host function letting a guest ask the runtime + // to verify one. No such import exists. It should be designed before a real + // key depends on it. See docs/STREAMING.md. +} + +message ServerMsg { + string session_id = 1; + uint64 seq = 2; + Kind kind = 3; + bytes payload = 4; +} diff --git a/examples/guest-grpc/src/framing.rs b/examples/guest-grpc/src/framing.rs new file mode 100644 index 0000000..dc191f8 --- /dev/null +++ b/examples/guest-grpc/src/framing.rs @@ -0,0 +1,147 @@ +//! gRPC's message framing, which is five bytes. +//! +//! One compression flag, then a big-endian `u32` length, then the message. +//! Frames do not align with HTTP/2 DATA frames in either direction — a frame +//! may arrive split across several reads, and several may arrive in one — so +//! this is the only place allowed to assume anything about where they begin. + +use bytes::{BufMut, Bytes, BytesMut}; + +/// Larger than any message this service has a use for. +/// +/// The ceiling has to live here. The runtime bounds an ordinary request body +/// because it hashes it, but a stream is never hashed and never buffered, so +/// nothing upstream is counting: the guest is the only thing between a client +/// and an allocation as large as it cares to claim. +pub const MAX_MESSAGE_BYTES: usize = 64 * 1024; + +#[derive(Debug, PartialEq, Eq)] +pub enum Malformed { + /// We advertise no compression, so a frame claiming it is a client bug. + Compressed, + TooLarge(usize), +} + +impl std::fmt::Display for Malformed { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Malformed::Compressed => write!(f, "compressed frames are not accepted"), + Malformed::TooLarge(n) => { + write!(f, "a {n} byte message exceeds the {MAX_MESSAGE_BYTES} byte limit") + } + } + } +} + +/// Wrap one encoded message for the wire. +pub fn frame(message: &[u8]) -> Bytes { + let mut out = BytesMut::with_capacity(5 + message.len()); + out.put_u8(0); + out.put_u32(message.len() as u32); + out.put_slice(message); + out.freeze() +} + +/// Reassembles messages from however the bytes happen to arrive. +#[derive(Debug, Default)] +pub struct Deframer { + buffer: BytesMut, +} + +impl Deframer { + pub fn push(&mut self, data: &[u8]) { + self.buffer.extend_from_slice(data); + } + + /// The next complete message, if one has fully arrived. + pub fn next(&mut self) -> Result, Malformed> { + if self.buffer.len() < 5 { + return Ok(None); + } + if self.buffer[0] != 0 { + return Err(Malformed::Compressed); + } + let len = u32::from_be_bytes(self.buffer[1..5].try_into().expect("five bytes")) as usize; + if len > MAX_MESSAGE_BYTES { + return Err(Malformed::TooLarge(len)); + } + if self.buffer.len() < 5 + len { + return Ok(None); + } + let _prefix = self.buffer.split_to(5); + Ok(Some(self.buffer.split_to(len).freeze())) + } + + /// Whether anything is stranded part-way through a frame. + /// + /// [`Deframer::next`] splits off the prefix and the payload together, so + /// after a complete message the buffer is genuinely empty. Anything left is + /// therefore the beginning of a frame whose rest never arrived — either a + /// short header or a header whose declared payload is incomplete — and both + /// are the same fact: the peer stopped mid-message. + /// + /// This is what a half-close has to consult. A stream that ends here has + /// not ended cleanly, however orderly the close looked at the HTTP layer. + pub fn is_empty(&self) -> bool { + self.buffer.is_empty() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_framed_message_deframes_to_itself() { + let mut d = Deframer::default(); + d.push(&frame(b"hello")); + assert_eq!(d.next().unwrap().as_deref(), Some(&b"hello"[..])); + assert_eq!(d.next().unwrap(), None); + } + + /// The case that makes this a deframer rather than a parser: a message + /// split across reads, which is the normal state of affairs on the wire. + #[test] + fn a_message_split_across_reads_is_reassembled() { + let whole = frame(b"split me"); + let mut d = Deframer::default(); + d.push(&whole[..3]); + assert_eq!(d.next().unwrap(), None, "three bytes is not a message"); + d.push(&whole[3..7]); + assert_eq!(d.next().unwrap(), None, "still short of the payload"); + d.push(&whole[7..]); + assert_eq!(d.next().unwrap().as_deref(), Some(&b"split me"[..])); + } + + #[test] + fn several_messages_in_one_read_all_come_out() { + let mut d = Deframer::default(); + let mut buf = Vec::new(); + buf.extend_from_slice(&frame(b"one")); + buf.extend_from_slice(&frame(b"two")); + d.push(&buf); + assert_eq!(d.next().unwrap().as_deref(), Some(&b"one"[..])); + assert_eq!(d.next().unwrap().as_deref(), Some(&b"two"[..])); + assert_eq!(d.next().unwrap(), None); + } + + #[test] + fn an_oversized_length_is_refused_before_it_is_allocated() { + let mut d = Deframer::default(); + let mut header = vec![0u8]; + header.extend_from_slice(&(MAX_MESSAGE_BYTES as u32 + 1).to_be_bytes()); + d.push(&header); + assert_eq!( + d.next(), + Err(Malformed::TooLarge(MAX_MESSAGE_BYTES + 1)), + "the length was believed before it was checked" + ); + } + + #[test] + fn a_compressed_frame_is_refused() { + let mut d = Deframer::default(); + d.push(&[1u8, 0, 0, 0, 1, b'x']); + assert_eq!(d.next(), Err(Malformed::Compressed)); + } +} diff --git a/examples/guest-grpc/src/main.rs b/examples/guest-grpc/src/main.rs new file mode 100644 index 0000000..aa01b6b --- /dev/null +++ b/examples/guest-grpc/src/main.rs @@ -0,0 +1,350 @@ +//! A gRPC service with a bidirectional streaming method, in a Wasm guest. +//! +//! ```text +//! client ──frame──▶ ┌──────────┐ ──frame──▶ client +//! client ◀─frame─── │ Session │ ◀─frame─── client +//! └──────────┘ +//! one task, both directions, neither waiting for the other +//! ``` +//! +//! # Why this is possible at all +//! +//! `wasi:http/incoming-handler` looks request/response, and half of it is: the +//! guest is called once and returns once. But `response-outparam.set` is +//! documented to *"allow execution to continue after the response has been +//! sent"*, and `incoming-request.consume` borrows rather than consumes — so +//! the request body and the response body are two independent resource trees +//! the guest may hold at the same time. The response head goes out first, and +//! everything after it is a conversation. +//! +//! [`Session`] is where that happens. It is an `http_body::Body` that owns the +//! *request* body: every time the host asks it for the next response frame it +//! first reads whatever the client has sent, answers it, and hands back the +//! answer. One task alternating, rather than two running — which is enough for +//! a request/response signing protocol and is worth naming as the limit it is. +//! +//! # What this is not +//! +//! There is no cryptography here. The messages are shaped like signing-session +//! traffic so the transport and the authorization can be settled before a key +//! depends on either, and `Sign` echoes rather than signs. + +mod framing; + +use std::collections::VecDeque; +use std::pin::Pin; +use std::task::{Context, Poll}; + +use bytes::Bytes; +use framing::{frame, Deframer}; +use http_body::{Body as HttpBody, Frame}; +use prost::Message as _; +use wstd::http::{Body, Error, HeaderMap, Request, Response, StatusCode}; + +// --- the wire format, mirroring proto/enclave/cosign/v1/session.proto ------- + +#[derive(Clone, Copy, Debug, PartialEq, Eq, prost::Enumeration)] +#[repr(i32)] +pub enum Kind { + Unspecified = 0, + Hello = 1, + Round = 2, + Finish = 3, +} + +#[derive(Clone, PartialEq, prost::Message)] +pub struct ClientMsg { + #[prost(string, tag = "1")] + pub session_id: String, + #[prost(uint64, tag = "2")] + pub seq: u64, + #[prost(enumeration = "Kind", tag = "3")] + pub kind: i32, + #[prost(bytes = "vec", tag = "4")] + pub payload: Vec, +} + +#[derive(Clone, PartialEq, prost::Message)] +pub struct ServerMsg { + #[prost(string, tag = "1")] + pub session_id: String, + #[prost(uint64, tag = "2")] + pub seq: u64, + #[prost(enumeration = "Kind", tag = "3")] + pub kind: i32, + #[prost(bytes = "vec", tag = "4")] + pub payload: Vec, +} + +// --- gRPC status codes this service uses ------------------------------------ + +const GRPC_OK: u32 = 0; +const GRPC_INVALID_ARGUMENT: u32 = 3; +const GRPC_PERMISSION_DENIED: u32 = 7; +const GRPC_UNIMPLEMENTED: u32 = 12; +const GRPC_UNAVAILABLE: u32 = 14; + +/// What the guest does with each message it receives. +/// +/// The variants past `Echo` exist to be tested against, the way `/hang` and +/// `/trickle` do in `examples/guest-http`: they are behaviours a host cannot +/// provoke from outside, so the guest has to offer them deliberately. +#[derive(Clone, Copy, PartialEq, Eq)] +enum Mode { + /// Answer every message as it arrives. The real shape. + Echo, + /// Answer once, then end with a non-zero status — so a test can read a + /// failure out of the trailers rather than out of the head. + Refuse, + /// Answer once, then burn CPU forever. The runtime's epoch watchdog must + /// still stop this, which is what keeps "a stream may run long" from + /// becoming "anything may run forever". + Spin, + /// Write without ever reading, so a client that does not drain finds out + /// that the guest stops rather than buffering. + Firehose, +} + +enum Phase { + Talking, + Trailers, + Done, +} + +struct Session { + /// The request body, held for the life of the response. This is the whole + /// trick, and it is specified behaviour rather than a trick: see the module + /// docs. + inbound: http_body_util::combinators::UnsyncBoxBody, + deframer: Deframer, + /// Answers waiting to go out. Bounded by what the client has sent, except + /// in `Firehose`, which is the point of that mode. + outbound: VecDeque, + mode: Mode, + phase: Phase, + status: (u32, String), + answered: u64, +} + +impl Session { + fn new( + inbound: http_body_util::combinators::UnsyncBoxBody, + mode: Mode, + ) -> Self { + let mut session = Session { + inbound, + deframer: Deframer::default(), + outbound: VecDeque::new(), + mode, + phase: Phase::Talking, + status: (GRPC_OK, String::new()), + answered: 0, + }; + if mode == Mode::Firehose { + // Queued up front and never read from the client, so the only + // thing regulating this is whether anybody is draining it. + for seq in 0..64 { + session.outbound.push_back(frame( + &ServerMsg { + session_id: "firehose".into(), + seq, + kind: Kind::Round as i32, + payload: vec![0xab; 1024], + } + .encode_to_vec(), + )); + } + } + session + } + + fn finish(&mut self, code: u32, message: &str) { + self.status = (code, message.to_string()); + self.phase = Phase::Trailers; + } + + fn trailers(&self) -> HeaderMap { + let mut trailers = HeaderMap::new(); + trailers.insert("grpc-status", self.status.0.to_string().parse().unwrap()); + if !self.status.1.is_empty() { + trailers.insert("grpc-message", self.status.1.parse().unwrap()); + } + trailers + } + + /// One answer for one message. + fn answer(&mut self, msg: ClientMsg) -> ServerMsg { + self.answered += 1; + ServerMsg { + session_id: msg.session_id, + seq: msg.seq, + kind: match msg.kind { + k if k == Kind::Hello as i32 => Kind::Hello as i32, + k if k == Kind::Finish as i32 => Kind::Finish as i32, + _ => Kind::Round as i32, + }, + // Echoed, not signed. A cosigner would do its round here — and + // would first need per-message approval, which nothing in this + // guest can obtain: verifying an assertion needs state that lives + // in the runtime, and there is no host function to ask through. + // See the `.proto` beside this file. + payload: msg.payload, + } + } +} + +impl HttpBody for Session { + type Data = Bytes; + type Error = Error; + + fn poll_frame( + self: Pin<&mut Self>, + cx: &mut Context<'_>, + ) -> Poll, Error>>> { + let this = self.get_mut(); + loop { + // Anything already answered goes out first. Each of these becomes + // one write on the response stream, and the host's outgoing buffer + // holds two chunks — so a client that is not reading stops this + // body being polled rather than letting it grow. + if let Some(ready) = this.outbound.pop_front() { + return Poll::Ready(Some(Ok(Frame::data(ready)))); + } + + match this.phase { + Phase::Trailers => { + this.phase = Phase::Done; + return Poll::Ready(Some(Ok(Frame::trailers(this.trailers())))); + } + Phase::Done => return Poll::Ready(None), + Phase::Talking => {} + } + + if this.mode == Mode::Firehose { + // Everything it had to say has now gone out. + this.finish(GRPC_OK, ""); + continue; + } + + if this.mode == Mode::Spin && this.answered > 0 { + // No await point, no I/O, forever. Only the epoch can end this. + #[allow(clippy::empty_loop)] + loop { + std::hint::spin_loop(); + } + } + + // Nothing to write, so read. This alternation *is* the concurrency + // model: one task, driven by the reactor's single `poll` over every + // pollable it holds. + match Pin::new(&mut this.inbound).poll_frame(cx) { + Poll::Pending => return Poll::Pending, + // The client half-closed — which is an orderly end only if it + // came between messages. A half-close with bytes still stranded + // in the deframer is a truncated frame, and reporting OK for it + // would tell the client its last message was received when it + // was not. The close looked clean at the HTTP layer; the gRPC + // stream did not end cleanly, and the trailers are the only + // place that difference can be said. + Poll::Ready(None) => { + if this.deframer.is_empty() { + this.finish(GRPC_OK, ""); + } else { + this.finish( + GRPC_INVALID_ARGUMENT, + "the stream ended part-way through a message", + ); + } + } + Poll::Ready(Some(Err(e))) => { + this.finish(GRPC_UNAVAILABLE, &e.to_string()); + } + Poll::Ready(Some(Ok(incoming))) => { + let Some(data) = incoming.data_ref() else { + // Request trailers. gRPC clients send none, and there + // is nothing this service would do with them. + continue; + }; + this.deframer.push(data); + loop { + match this.deframer.next() { + Ok(Some(message)) => match ClientMsg::decode(message) { + Ok(msg) => { + if this.mode == Mode::Refuse { + let reply = this.answer(msg); + this.outbound.push_back(frame(&reply.encode_to_vec())); + this.finish( + GRPC_PERMISSION_DENIED, + "this session is not authorized to sign", + ); + break; + } + let reply = this.answer(msg); + this.outbound.push_back(frame(&reply.encode_to_vec())); + } + Err(e) => { + this.finish( + GRPC_INVALID_ARGUMENT, + &format!("undecodable ClientMsg: {e}"), + ); + break; + } + }, + Ok(None) => break, + Err(malformed) => { + this.finish(GRPC_INVALID_ARGUMENT, &malformed.to_string()); + break; + } + } + } + } + } + } + } +} + +/// gRPC is always HTTP 200; the real status is in the trailers. +fn grpc_response(session: Session) -> Response { + Response::builder() + .status(StatusCode::OK) + .header("content-type", "application/grpc+proto") + // Load-bearing on HTTP/1.1: hyper's encoder drops trailers outright + // unless the response declares which ones it will send. Harmless on + // HTTP/2, where trailers need no announcement. + .header("trailer", "grpc-status, grpc-message") + .body(Body::from_http_body(session)) + .expect("response is well formed") +} + +/// An unknown method is a gRPC status, not an HTTP one — a client reading only +/// the head would otherwise see a perfectly successful call. +fn unimplemented() -> Response { + use http_body_util::BodyExt as _; + let mut trailers = HeaderMap::new(); + trailers.insert("grpc-status", GRPC_UNIMPLEMENTED.to_string().parse().unwrap()); + trailers.insert("grpc-message", "no such method".parse().unwrap()); + Response::builder() + .status(StatusCode::OK) + .header("content-type", "application/grpc+proto") + .header("trailer", "grpc-status, grpc-message") + .body(Body::from_http_body( + http_body_util::Empty::::new() + .map_err(|e: std::convert::Infallible| match e {}) + .with_trailers(async move { Some(Ok::<_, Error>(trailers)) }), + )) + .expect("response is well formed") +} + +#[wstd::http_server] +async fn main(req: Request) -> Result, Error> { + let path = req.uri().path().to_string(); + let mode = match path.as_str() { + "/enclave.cosign.v1.SigningSession/Sign" => Mode::Echo, + "/enclave.cosign.v1.SigningSession/Refuse" => Mode::Refuse, + "/enclave.cosign.v1.SigningSession/Spin" => Mode::Spin, + "/enclave.cosign.v1.SigningSession/Firehose" => Mode::Firehose, + _ => return Ok(unimplemented()), + }; + let inbound = req.into_body().into_boxed_body(); + Ok(grpc_response(Session::new(inbound, mode))) +} diff --git a/examples/guest-http/Cargo.lock b/examples/guest-http/Cargo.lock new file mode 100644 index 0000000..1e1a702 --- /dev/null +++ b/examples/guest-http/Cargo.lock @@ -0,0 +1,527 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + +[[package]] +name = "async-task" +version = "4.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b75356056920673b02621b35afd0f7dda9306d03c79a30f5c56c44cf256e3de" + +[[package]] +name = "bitflags" +version = "2.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" + +[[package]] +name = "bytes" +version = "1.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" + +[[package]] +name = "cfg-if" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" + +[[package]] +name = "equivalent" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" + +[[package]] +name = "fastrand" +version = "1.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e51093e27b0797c359783294ca4f0a911c270184cb10f85783b118614a1501be" +dependencies = [ + "instant", +] + +[[package]] +name = "fastrand" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" + +[[package]] +name = "foldhash" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2" + +[[package]] +name = "futures-core" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" + +[[package]] +name = "futures-io" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4577ecaa3c4f96589d473f679a71b596316f6641bc350038b962a5daf0085d7a" + +[[package]] +name = "futures-lite" +version = "1.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "49a9d51ce47660b1e808d3c990b4709f2f415d928835a17dfd16991515c46bce" +dependencies = [ + "fastrand 1.9.0", + "futures-core", + "futures-io", + "memchr", + "parking", + "pin-project-lite", + "waker-fn", +] + +[[package]] +name = "futures-lite" +version = "2.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f78e10609fe0e0b3f4157ffab1876319b5b0db102a2c60dc4626306dc46b44ad" +dependencies = [ + "fastrand 2.5.0", + "futures-core", + "futures-io", + "parking", + "pin-project-lite", +] + +[[package]] +name = "guest-http" +version = "0.1.0" +dependencies = [ + "futures-lite 2.6.1", + "wit-bindgen 0.51.0", + "wstd", +] + +[[package]] +name = "hashbrown" +version = "0.15.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" +dependencies = [ + "foldhash", +] + +[[package]] +name = "hashbrown" +version = "0.17.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" + +[[package]] +name = "heck" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" + +[[package]] +name = "http" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" +dependencies = [ + "bytes", + "itoa", +] + +[[package]] +name = "http-body" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" +dependencies = [ + "bytes", + "http", +] + +[[package]] +name = "http-body-util" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2" +dependencies = [ + "bytes", + "futures-core", + "http", + "http-body", + "pin-project-lite", +] + +[[package]] +name = "id-arena" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954" + +[[package]] +name = "indexmap" +version = "2.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d466e9454f08e4a911e14806c24e16fba1b4c121d1ea474396f396069cf949d9" +dependencies = [ + "equivalent", + "hashbrown 0.17.1", + "serde", + "serde_core", +] + +[[package]] +name = "instant" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e0242819d153cba4b4b05a5a8f2a7e9bbf97b6055b2a002b395c96b5ff3c0222" +dependencies = [ + "cfg-if", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "leb128fmt" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09edd9e8b54e49e587e4f6295a7d29c3ea94d469cb40ab8ca70b288248a81db2" + +[[package]] +name = "log" +version = "0.4.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "parking" +version = "2.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f38d5652c16fde515bb1ecef450ab0f6a219d619a7274976324d5e377f7dceba" + +[[package]] +name = "pin-project-lite" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" + +[[package]] +name = "prettyplease" +version = "0.2.37" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "479ca8adacdd7ce8f1fb39ce9ecccbfe93a3f1344b3d0d97f20bc0196208f62b" +dependencies = [ + "proc-macro2", + "syn 2.0.119", +] + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "semver" +version = "1.0.28" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "slab" +version = "0.4.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" + +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "unicode-ident" +version = "1.0.24" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" + +[[package]] +name = "unicode-xid" +version = "0.2.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ebc1c04c71510c7f702b52b7c350734c9ff1295c464a03335b00bb84fc54f853" + +[[package]] +name = "waker-fn" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "317211a0dc0ceedd78fb2ca9a44aed3d7b9b26f81870d485c07122b4350673b7" + +[[package]] +name = "wasip2" +version = "1.0.4+wasi-0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" +dependencies = [ + "wit-bindgen 0.57.1", +] + +[[package]] +name = "wasm-encoder" +version = "0.244.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "990065f2fe63003fe337b932cfb5e3b80e0b4d0f5ff650e6985b1048f62c8319" +dependencies = [ + "leb128fmt", + "wasmparser", +] + +[[package]] +name = "wasm-metadata" +version = "0.244.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb0e353e6a2fbdc176932bbaab493762eb1255a7900fe0fea1a2f96c296cc909" +dependencies = [ + "anyhow", + "indexmap", + "wasm-encoder", + "wasmparser", +] + +[[package]] +name = "wasmparser" +version = "0.244.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe" +dependencies = [ + "bitflags", + "hashbrown 0.15.5", + "indexmap", + "semver", +] + +[[package]] +name = "wit-bindgen" +version = "0.51.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d7249219f66ced02969388cf2bb044a09756a083d0fab1e566056b04d9fbcaa5" +dependencies = [ + "bitflags", + "wit-bindgen-rust-macro", +] + +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" +dependencies = [ + "bitflags", +] + +[[package]] +name = "wit-bindgen-core" +version = "0.51.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ea61de684c3ea68cb082b7a88508a8b27fcc8b797d738bfc99a82facf1d752dc" +dependencies = [ + "anyhow", + "heck", + "wit-parser", +] + +[[package]] +name = "wit-bindgen-rust" +version = "0.51.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b7c566e0f4b284dd6561c786d9cb0142da491f46a9fbed79ea69cdad5db17f21" +dependencies = [ + "anyhow", + "heck", + "indexmap", + "prettyplease", + "syn 2.0.119", + "wasm-metadata", + "wit-bindgen-core", + "wit-component", +] + +[[package]] +name = "wit-bindgen-rust-macro" +version = "0.51.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c0f9bfd77e6a48eccf51359e3ae77140a7f50b1e2ebfe62422d8afdaffab17a" +dependencies = [ + "anyhow", + "prettyplease", + "proc-macro2", + "quote", + "syn 2.0.119", + "wit-bindgen-core", + "wit-bindgen-rust", +] + +[[package]] +name = "wit-component" +version = "0.244.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2" +dependencies = [ + "anyhow", + "bitflags", + "indexmap", + "log", + "serde", + "serde_derive", + "serde_json", + "wasm-encoder", + "wasm-metadata", + "wasmparser", + "wit-parser", +] + +[[package]] +name = "wit-parser" +version = "0.244.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ecc8ac4bc1dc3381b7f59c34f00b67e18f910c2c0f50015669dde7def656a736" +dependencies = [ + "anyhow", + "id-arena", + "indexmap", + "log", + "semver", + "serde", + "serde_derive", + "serde_json", + "unicode-xid", + "wasmparser", +] + +[[package]] +name = "wstd" +version = "0.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29b52936db10a79bb724dadd1c2c3aac958e8229dcb1f1c7f2b7044ca9fc6a3a" +dependencies = [ + "anyhow", + "async-task", + "bytes", + "futures-lite 1.13.0", + "http", + "http-body", + "http-body-util", + "itoa", + "pin-project-lite", + "serde", + "serde_json", + "slab", + "wasip2", + "wstd-macro", +] + +[[package]] +name = "wstd-macro" +version = "0.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "153db9b65508bf6c2efe26617169840c03671ba5719a35868db7682a5261f7d4" +dependencies = [ + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/examples/guest-http/Cargo.toml b/examples/guest-http/Cargo.toml new file mode 100644 index 0000000..42e1e3c --- /dev/null +++ b/examples/guest-http/Cargo.toml @@ -0,0 +1,29 @@ +[workspace] + +[package] +name = "guest-http" +version = "0.1.0" +edition = "2021" +publish = false +description = "wasm32-wasip2 guest serving wasi:http/incoming-handler over the block store" + +[[bin]] +name = "guest-http" +path = "src/main.rs" + +[dependencies] +# `wstd` supplies the `http_server` attribute, which is what turns this binary +# into a component exporting `wasi:http/incoming-handler`. Everything else is +# `std`: file access goes through `wasi:filesystem`, which the runtime backs +# with the block store. +wstd = "0.6" +# For `/trickle`, whose body is a stream rather than a buffer: `wstd`'s +# `Body::from_stream` takes a `futures_lite::Stream`, and `unfold` is the +# shortest way to write one without a generator. +futures-lite = "2" +wit-bindgen = "0.51" + +[profile.release] +opt-level = "s" +lto = true +codegen-units = 1 diff --git a/examples/guest-http/src/main.rs b/examples/guest-http/src/main.rs new file mode 100644 index 0000000..87b83cf --- /dev/null +++ b/examples/guest-http/src/main.rs @@ -0,0 +1,459 @@ +//! A guest that answers HTTP requests out of the Merkle-anchored filesystem. +//! +//! This is the end-to-end signal for the serving path, and it is deliberately +//! stateful: `/counter` reads a file, increments it and writes it back, so a +//! second request proves the runtime carried a *committed* filesystem across +//! requests rather than handing each instance a fresh view. A stateless +//! handler would pass even if every request got its own empty store. +//! +//! What it never does is as informative as what it does. There is no listener, +//! no socket, no certificate and no outbound request anywhere in this file — +//! the runtime terminates TLS and hands over a parsed request. The guest's +//! entire view of the network is the `Request` argument. +//! +//! ```console +//! $ curl localhost:8080/ # what this guest is +//! $ curl localhost:8080/counter # increments, persists +//! $ curl localhost:8080/env # what the environment policy let through +//! $ curl -X POST --data-binary @f localhost:8080/files/notes.txt +//! $ curl localhost:8080/files/notes.txt +//! ``` + +use std::cell::Cell; +use std::fmt::Write as _; +use std::fs; +use std::io::Write as _; +use std::path::{Component, Path, PathBuf}; + +use wstd::http::{Body, Error, Request, Response, StatusCode}; + +mod bindings { + wit_bindgen::generate!({ path: "wit", world: "app", generate_all }); +} +struct Background; +impl bindings::Guest for Background { + fn run_task(task_id: String, payload: Vec) -> Result, String> { + // This example's effect is one durable file per occurrence. Retrying + // the same occurrence reads that result instead of repeating its work. + let dir = format!("{STATE_DIR}/tasks"); + fs::create_dir_all(&dir).map_err(|e| e.to_string())?; + let path = format!("{dir}/{task_id}"); + match fs::read(&path) { + Ok(bytes) => return Ok(bytes), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => {} + Err(e) => return Err(e.to_string()), + } + if payload == b"fail" { + return Err("example task failure".into()); + } + if payload == b"spin" { + loop { + std::hint::spin_loop(); + } + } + if payload == b"check-authority" { + if bindings::enclave::tasks::queue::enqueue("unauthorized", b"nested", 0, None).is_ok() + { + return Err("background invocation granted itself work".into()); + } + // Enrolling a device is standing authority to reach somebody, so + // background work must not be able to do it. + // + // Only the refusal is asserted here. The other half — that a wake + // *is* allowed from background — cannot be checked in the same + // breath, because `wake` also fails when no sender is configured, + // which is how most tests run. It is proven where it actually + // matters instead: the wake raised at the end of this function, in + // leg 6b of the QEMU harness. + if bindings::enclave::notify::notify::register_device(&"z".repeat(64)).is_ok() { + return Err("background invocation enrolled a device".into()); + } + } + let temp = format!("{path}.tmp"); + let mut file = fs::File::create(&temp).map_err(|e| e.to_string())?; + file.write_all(&payload).map_err(|e| e.to_string())?; + file.sync_all().map_err(|e| e.to_string())?; + drop(file); + fs::rename(&temp, &path).map_err(|e| e.to_string())?; + + // Tell the owner there is something to come back for. The run id is + // `::` and only the first part is a label, + // so that is what travels — and it is a reference, not a message: the + // app fetches the detail over its own attested connection. + // + // Best effort on purpose. A wake that cannot be queued must not fail + // work that has already been committed. + let reference = task_id.split(':').next().unwrap_or("task"); + if let Err(e) = bindings::enclave::notify::notify::wake("task-done", Some(reference)) { + eprintln!("could not raise a wake for {reference}: {e}"); + } + Ok(payload) + } +} +bindings::export!(Background with_types_in bindings); + +/// Where this guest keeps its state. A directory rather than the root, so it +/// is obvious in a bucket listing which objects came from the example. +const STATE_DIR: &str = "/http-example"; + +thread_local! { + /// A counter that touches no storage at all. + /// + /// `/counter` proves the *filesystem* carried state between requests. This + /// proves the opposite, and it is the more important of the two: a fresh + /// instance has fresh memory, so `/memory` must answer 1 forever. Any other + /// answer means two requests reached the same instance, and the boundary + /// between two clients has quietly moved out of the runtime and into this + /// file. + static IN_MEMORY: Cell = const { Cell::new(0) }; +} + +#[wstd::http_server] +async fn main(mut req: Request) -> Result, Error> { + let path = req.uri().path().to_string(); + let method = req.method().clone(); + + let result = match (method.as_str(), path.as_str()) { + ("GET", "/") => Ok(text(StatusCode::OK, banner())), + ("GET", "/counter") => counter().map(|n| text(StatusCode::OK, format!("{n}\n"))), + ("GET", "/memory") => Ok(text( + StatusCode::OK, + format!( + "{}\n", + IN_MEMORY.with(|c| { + c.set(c.get() + 1); + c.get() + }) + ), + )), + ("POST", p) if p.starts_with("/tasks/") => { + let id = &p[7..]; + let time_header = |name: &str| -> Result, String> { + req.headers() + .get(name) + .map(|value| { + value + .to_str() + .ok() + .and_then(|value| value.parse().ok()) + .ok_or_else(|| format!("{name} must be an unsigned millisecond value")) + }) + .transpose() + }; + let (at, interval) = match ( + time_header("x-task-run-at"), + time_header("x-task-interval-ms"), + ) { + (Ok(at), Ok(interval)) => (at.unwrap_or(0), interval), + (Err(error), _) | (_, Err(error)) => { + return Ok(text(StatusCode::BAD_REQUEST, error)) + } + }; + let payload = req.body_mut().bytes_contents().await?; + Ok( + match bindings::enclave::tasks::queue::enqueue(id, &payload, at, interval) { + Ok(()) => text(StatusCode::ACCEPTED, id.to_string()), + Err(e) => text(StatusCode::BAD_REQUEST, e), + }, + ) + } + ("GET", p) if p.starts_with("/tasks/") => { + Ok(match bindings::enclave::tasks::queue::status(&p[7..]) { + Ok(record) => text(StatusCode::OK, record), + Err(e) => text(StatusCode::NOT_FOUND, e), + }) + } + ("DELETE", p) if p.starts_with("/tasks/") => { + let id = &p[7..]; + let result = match id.strip_suffix("/forget") { + Some(id) => bindings::enclave::tasks::queue::forget(id), + None => bindings::enclave::tasks::queue::cancel(id), + }; + Ok(match result { + Ok(()) => text(StatusCode::OK, "ok".to_string()), + Err(e) => text(StatusCode::BAD_REQUEST, e), + }) + } + // Enrolling is an ordinary signed interaction, so it needs no route of + // the runtime's own: the gate already stands in front of this one. + ("POST", "/devices") => { + let body = req.body_mut().bytes_contents().await?; + let token = String::from_utf8_lossy(&body); + Ok(match bindings::enclave::notify::notify::register_device(token.trim()) { + Ok(()) => text(StatusCode::OK, "enrolled".to_string()), + Err(e) => text(StatusCode::BAD_REQUEST, e), + }) + } + // A count, never the tokens: one is a capability to wake that device + // from anywhere, and the runtime does not hand them back. + ("GET", "/devices") => Ok(match bindings::enclave::notify::notify::devices() { + Ok(count) => text(StatusCode::OK, format!("{count}\n")), + Err(e) => text(StatusCode::BAD_REQUEST, e), + }), + ("DELETE", "/devices") => { + let body = req.body_mut().bytes_contents().await?; + let token = String::from_utf8_lossy(&body); + Ok(match bindings::enclave::notify::notify::forget_device(token.trim()) { + Ok(()) => text(StatusCode::OK, "forgotten".to_string()), + Err(e) => text(StatusCode::BAD_REQUEST, e), + }) + } + ("GET", "/env") => Ok(text(StatusCode::OK, environment())), + // Whatever the runtime says about the caller. A guest can only ever + // read this header, never write it, and the runtime overwrites it on + // every request — so what arrives here is the runtime's word, not the + // client's. + ("GET", "/whoami") => Ok(text( + StatusCode::OK, + match req.headers().get("x-enclave-tenant") { + Some(v) => format!("{}\n", v.to_str().unwrap_or("(not utf-8)")), + None => "(anonymous)\n".to_string(), + }, + )), + ("POST", p) | ("PUT", p) if p.starts_with("/files/") => { + let body = req.body_mut().bytes_contents().await?; + write_file(&p["/files/".len()..], &body) + } + ("GET", p) if p.starts_with("/files/") => read_file(&p["/files/".len()..]), + // A guest that does **not** validate the path, on purpose. + // + // `/files/` sanitises; this deliberately does not, so a test can point + // it at `../../` or `/tenants/someone-else` and see what the *runtime* + // does. That is the whole question for a tenant: separation has to be + // the capability layer's, not this file's, and the only way to show it + // is with a guest that is not helping. + ("GET", p) if p.starts_with("/escape/") => { + let target = &p["/escape/".len()..]; + match fs::read(target) { + Ok(bytes) => Ok(text( + StatusCode::OK, + format!("read {} bytes from {target}\n", bytes.len()), + )), + Err(e) => Ok(text( + StatusCode::NOT_FOUND, + format!("refused {target}: {e}\n"), + )), + } + } + // Known values on both streams, for the runtime's guest-logging + // tests. Every case the line framer has to get right is here, written + // the way a guest would actually write it: a line built from two + // `print!`s, a blank line, CRLF, bytes that are not UTF-8, something + // on stderr, and a final line with no terminator at all. + // + // Flushed explicitly. Rust's stdout is line buffered and a component's + // `handle` returning is not process exit, so without this the tail + // would sit in the guest's own buffer and never reach the host. + ("GET", "/log") => { + use std::io::Write; + let mut out = std::io::stdout(); + let _ = out.write_all(b"first "); + let _ = out.write_all(b"line\n"); + let _ = out.write_all(b"\n"); + let _ = out.write_all(b"windows\r\n"); + let _ = out.write_all(b"invalid \xff\xfe bytes\n"); + let _ = out.write_all(b"no trailing newline"); + let _ = out.flush(); + + let mut err = std::io::stderr(); + let _ = err.write_all(b"on stderr\n"); + let _ = err.flush(); + + Ok(text(StatusCode::OK, "logged\n".to_string())) + } + // A guest doing its best to own the runtime's proof header. The + // runtime must overwrite all of these, or a guest could tell a client + // whatever it liked about the enclave it is running in. + ("GET", "/forge-attestation") => { + let mut response = Response::builder().status(StatusCode::OK); + for forged in ["forged-one", "forged-two", "forged-three"] { + response = response.header("x-enclave-attestation", forged); + } + Ok(response + .header("cache-control", "public, max-age=3600") + .header("content-type", "text/plain; charset=utf-8") + .body("tried\n".to_string().into()) + .expect("response is well formed")) + } + // A body that arrives in pieces, with a pause between them. The + // runtime must send the response head — attestation header and all — + // as soon as the guest sets it, and not wait for the body to finish. + // Without a deliberate pause here that property is a race the test + // could win by accident. + ("GET", "/trickle") => { + let stream = futures_lite::stream::unfold(0u32, |n| async move { + if n >= 3 { + return None; + } + if n > 0 { + wstd::task::sleep(std::time::Duration::from_millis(150).into()).await; + } + Some((format!("chunk-{n}\n"), n + 1)) + }); + Ok(Response::builder() + .status(StatusCode::OK) + .header("content-type", "text/plain; charset=utf-8") + .body(Body::from_stream(stream)) + .expect("response is well formed")) + } + // A guest that never returns and never sets a response. Deliberately + // here rather than in a test fixture: it is the one behaviour a host + // cannot provoke from the outside, and without it the runtime's + // watchdog has nothing to be tested against. + // Ten chunks over ~2s: long enough that any sane head timeout has + // passed several times over while the stream is still healthy. The + // watchdog must judge this by the bytes it is moving, not by how long + // it has been running — `/hang` is the case that must still die. + ("GET", "/slow-stream") => { + let stream = futures_lite::stream::unfold(0u32, |n| async move { + if n >= 10 { + return None; + } + wstd::task::sleep(std::time::Duration::from_millis(200).into()).await; + Some((format!("tick-{n}\n"), n + 1)) + }); + Ok(Response::builder() + .status(StatusCode::OK) + .header("content-type", "text/plain; charset=utf-8") + .body(Body::from_stream(stream)) + .expect("response is well formed")) + } + // Parked in a *host* call, not spinning. The epoch cannot see this — + // there is no wasm executing to interrupt — so it is the case + // `await_head`'s abort exists for, and the only way to reach that path + // from a test. `/hang` is its opposite: wasm the epoch must stop. + ("GET", "/park") => { + wstd::task::sleep(std::time::Duration::from_secs(30).into()).await; + Ok(text(StatusCode::OK, "woke\n".to_string())) + } + ("GET", "/hang") => + { + #[allow(clippy::empty_loop)] + loop { + std::hint::spin_loop(); + } + } + _ => Ok(text( + StatusCode::NOT_FOUND, + format!("no route for {method} {path}\n"), + )), + }; + + // A guest failure is a 500 with the reason, not a trap. Trapping would + // take down the instance and tell the client nothing, and inside an + // enclave the console is the only other place the reason could go. + Ok(result.unwrap_or_else(|e| text(StatusCode::INTERNAL_SERVER_ERROR, format!("error: {e}\n")))) +} + +fn text(status: StatusCode, body: String) -> Response { + Response::builder() + .status(status) + .header("content-type", "text/plain; charset=utf-8") + .body(body.into()) + .expect("response is well formed") +} + +fn banner() -> String { + let mut out = String::from("guest-http on a Merkle-anchored filesystem\n\n"); + out.push_str("GET /counter increment and return a persisted counter\n"); + out.push_str("GET /env environment the runtime policy allowed\n"); + out.push_str("POST /files/ write a file\n"); + out.push_str("GET /files/ read it back\n"); + out.push_str("GET /trickle a body sent in pieces, with pauses\n\n"); + match fs::read_dir(STATE_DIR) { + Ok(entries) => { + let _ = writeln!(out, "files: {}", entries.count()); + } + Err(_) => out.push_str("files: none yet\n"), + } + out +} + +/// Read-modify-write against the block store. +/// +/// The proof that matters: run it twice and the number goes up. That can only +/// happen if the first request's write was committed and the second request's +/// fresh instance read the committed state. +fn counter() -> Result { + fs::create_dir_all(STATE_DIR).map_err(|e| format!("creating {STATE_DIR}: {e}"))?; + let path = format!("{STATE_DIR}/counter"); + + let current: u64 = match fs::read_to_string(&path) { + Ok(s) => s.trim().parse().unwrap_or(0), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => 0, + Err(e) => return Err(format!("reading {path}: {e}")), + }; + let next = current + 1; + + // `sync_all` rather than relying on drop: the runtime commits a + // transaction on flush, and a response that reports a number the store has + // not accepted yet would be a lie the next request exposes. + let mut f = fs::File::create(&path).map_err(|e| format!("creating {path}: {e}"))?; + f.write_all(format!("{next}\n").as_bytes()) + .map_err(|e| format!("writing {path}: {e}"))?; + f.sync_all().map_err(|e| format!("syncing {path}: {e}"))?; + Ok(next) +} + +fn environment() -> String { + let mut vars: Vec<_> = std::env::vars().collect(); + vars.sort(); + if vars.is_empty() { + return "(the runtime passed no variables)\n".to_string(); + } + let mut out = String::new(); + for (k, v) in vars { + let _ = writeln!(out, "{k}={v}"); + } + out +} + +fn write_file(name: &str, body: &[u8]) -> Result, String> { + let path = safe_path(name)?; + fs::create_dir_all(STATE_DIR).map_err(|e| format!("creating {STATE_DIR}: {e}"))?; + let mut f = fs::File::create(&path).map_err(|e| format!("creating {}: {e}", path.display()))?; + f.write_all(body) + .map_err(|e| format!("writing {}: {e}", path.display()))?; + f.sync_all() + .map_err(|e| format!("syncing {}: {e}", path.display()))?; + Ok(text( + StatusCode::CREATED, + format!("wrote {} bytes to {}\n", body.len(), path.display()), + )) +} + +fn read_file(name: &str) -> Result, String> { + let path = safe_path(name)?; + match fs::read(&path) { + Ok(bytes) => Ok(Response::builder() + .status(StatusCode::OK) + .header("content-type", "application/octet-stream") + .body(bytes.into()) + .expect("response is well formed")), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(text( + StatusCode::NOT_FOUND, + format!("no such file: {name}\n"), + )), + Err(e) => Err(format!("reading {}: {e}", path.display())), + } +} + +/// Keep a request-supplied name inside [`STATE_DIR`]. +/// +/// The preopen the runtime grants is the filesystem *root*, so `..` in a path +/// from a client would escape this guest's directory and reach anything else +/// stored in the same filesystem. WASI's capability model stops a guest +/// leaving its preopen; it does not stop a guest wandering around inside it. +fn safe_path(name: &str) -> Result { + if name.is_empty() { + return Err("empty file name".into()); + } + let candidate = Path::new(name); + if candidate + .components() + .any(|c| !matches!(c, Component::Normal(_))) + { + return Err(format!("rejected path {name:?}: must be a plain file name")); + } + Ok(Path::new(STATE_DIR).join(candidate)) +} diff --git a/examples/guest-http/wit/app.wit b/examples/guest-http/wit/app.wit new file mode 100644 index 0000000..8c2ed24 --- /dev/null +++ b/examples/guest-http/wit/app.wit @@ -0,0 +1,13 @@ +package guest:http@0.1.0; + +/// What this example needs from the runtime. +/// +/// Its own world, rather than an edited copy of `enclave:tasks`'s: the vendored +/// `tasks.wit` and `notify.wit` beside it are byte-for-byte copies of the +/// canonical ones under `wit/`, and `scripts/wit-drift.sh` checks that they stay +/// that way. A guest that wants two capabilities composes them here instead of +/// editing either. +world app { + include enclave:tasks/background@0.1.0; + import enclave:notify/notify@0.1.0; +} diff --git a/examples/guest-http/wit/deps/notify/notify.wit b/examples/guest-http/wit/deps/notify/notify.wit new file mode 100644 index 0000000..0edc9f8 --- /dev/null +++ b/examples/guest-http/wit/deps/notify/notify.wit @@ -0,0 +1,40 @@ +package enclave:notify@0.1.0; + +/// Waking a tenant's devices. Calls are bound to the current tenant by the +/// runtime; no function here accepts a tenant id. +/// +/// **Nothing here carries text a person will read.** The runtime sends a +/// data-only message containing an opaque category and an optional +/// tenant-local reference, and nothing else. That payload crosses the parent +/// instance and Google, which are the two parties this design excludes from a +/// tenant's data — so the app wakes and fetches the detail over the attested +/// channel. See docs/NOTIFICATIONS.md. +interface notify { + /// Enrol an FCM registration token for this tenant. + /// + /// **Interactive only.** Enrolling grants a standing ability to reach a + /// device, and background work cannot grant itself that — the same rule + /// enclave:tasks/queue applies to enqueue. Enrolling a token this tenant + /// already has is success and changes nothing. + /// + /// 32-512 characters of [A-Za-z0-9_:.-]. + register-device: func(token: string) -> result<_, string>; + + /// Interactive only. Forgetting a token this tenant does not have is + /// success: a client pruned while it was offline is not wrong to ask. + forget-device: func(token: string) -> result<_, string>; + + /// How many devices this tenant has enrolled. Never the tokens themselves. + devices: func() -> result; + + /// Queue a wake signal to every device this tenant has enrolled. + /// + /// Allowed from background work, and that is the point: a finished task + /// telling its owner to come and look is the primary use. It authorizes + /// nothing — it spends an enrolment an interactive call already made. + /// + /// `category` and `reference` are opaque labels: 1-64 characters of + /// [A-Za-z0-9_-]. Not prose, not a message, not a title. Returns once the + /// signal is queued; delivery is best effort and unacknowledged. + wake: func(category: string, reference: option) -> result<_, string>; +} diff --git a/examples/guest-http/wit/deps/tasks/tasks.wit b/examples/guest-http/wit/deps/tasks/tasks.wit new file mode 100644 index 0000000..0b31db7 --- /dev/null +++ b/examples/guest-http/wit/deps/tasks/tasks.wit @@ -0,0 +1,31 @@ +package enclave:tasks@0.1.0; + +/// Calls are bound to the current tenant by the runtime. Mutations require +/// an interactive invocation; a background invocation cannot grant itself work. +interface queue { + /// id is a tenant-local idempotency key (ASCII letters, digits, '-' or '_'). + /// run-at is Unix milliseconds. interval-ms, when present, is >= 1000. + enqueue: func(id: string, payload: list, run-at: u64, interval-ms: option) -> result<_, string>; + /// A JSON task record, including state, attempt count and result bytes. + status: func(id: string) -> result; + cancel: func(id: string) -> result<_, string>; + /// Remove a terminal record to release quota. Never removes running work. + forget: func(id: string) -> result<_, string>; +} + +world background { + import queue; + /// task-id is a run id, not the id the task was enqueued under. Its form + /// is `::`: the enqueued id, 32 lowercase hex + /// digits minted at enqueue, and a decimal count of recurring occurrences + /// starting at 0. An id cannot contain ':', so the enqueued id is + /// everything before the first one. + /// + /// It remains stable across retries of the same occurrence, and changes + /// for the next occurrence of a recurring task or for a task enqueued again + /// under the same id after `forget`. Deduplicate effects by the whole run + /// id; look up application state by the enqueued id. + /// + /// A successful result is persisted for the client; errors are retried. + export run-task: func(task-id: string, payload: list) -> result, string>; +} diff --git a/examples/guest-sqlite/Cargo.lock b/examples/guest-sqlite/Cargo.lock new file mode 100644 index 0000000..d8c76e0 --- /dev/null +++ b/examples/guest-sqlite/Cargo.lock @@ -0,0 +1,438 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "ahash" +version = "0.8.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a15f179cd60c4584b8a8c596927aadc462e27f2ca70c04e0071964a73ba7a75" +dependencies = [ + "cfg-if", + "once_cell", + "version_check", + "zerocopy", +] + +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + +[[package]] +name = "async-task" +version = "4.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b75356056920673b02621b35afd0f7dda9306d03c79a30f5c56c44cf256e3de" + +[[package]] +name = "bitflags" +version = "2.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" + +[[package]] +name = "bytes" +version = "1.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" + +[[package]] +name = "cc" +version = "1.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5d262e149917187838d5b42777c8253bcb64500067342904e7d429499a6f277e" +dependencies = [ + "find-msvc-tools", + "shlex", +] + +[[package]] +name = "cfg-if" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" + +[[package]] +name = "fallible-iterator" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" + +[[package]] +name = "fallible-streaming-iterator" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" + +[[package]] +name = "fastrand" +version = "1.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e51093e27b0797c359783294ca4f0a911c270184cb10f85783b118614a1501be" +dependencies = [ + "instant", +] + +[[package]] +name = "find-msvc-tools" +version = "0.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "26b73573e6edcd2af0cdf47bd6cb58f0b3839491263c314eaad1ccf24430e1de" + +[[package]] +name = "futures-core" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" + +[[package]] +name = "futures-io" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" + +[[package]] +name = "futures-lite" +version = "1.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "49a9d51ce47660b1e808d3c990b4709f2f415d928835a17dfd16991515c46bce" +dependencies = [ + "fastrand", + "futures-core", + "futures-io", + "memchr", + "parking", + "pin-project-lite", + "waker-fn", +] + +[[package]] +name = "guest-sqlite" +version = "0.1.0" +dependencies = [ + "rusqlite", + "wstd", +] + +[[package]] +name = "hashbrown" +version = "0.14.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e5274423e17b7c9fc20b6e7e208532f9b19825d82dfd615708b70edd83df41f1" +dependencies = [ + "ahash", +] + +[[package]] +name = "hashlink" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ba4ff7128dee98c7dc9794b6a411377e1404dba1c97deb8d1a55297bd25d8af" +dependencies = [ + "hashbrown", +] + +[[package]] +name = "http" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" +dependencies = [ + "bytes", + "itoa", +] + +[[package]] +name = "http-body" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" +dependencies = [ + "bytes", + "http", +] + +[[package]] +name = "http-body-util" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" +dependencies = [ + "bytes", + "futures-core", + "http", + "http-body", + "pin-project-lite", +] + +[[package]] +name = "instant" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e0242819d153cba4b4b05a5a8f2a7e9bbf97b6055b2a002b395c96b5ff3c0222" +dependencies = [ + "cfg-if", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "libsqlite3-sys" +version = "0.30.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2e99fb7a497b1e3339bc746195567ed8d3e24945ecd636e3619d20b9de9e9149" +dependencies = [ + "cc", + "pkg-config", + "vcpkg", +] + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + +[[package]] +name = "parking" +version = "2.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f38d5652c16fde515bb1ecef450ab0f6a219d619a7274976324d5e377f7dceba" + +[[package]] +name = "pin-project-lite" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" + +[[package]] +name = "pkg-config" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "rusqlite" +version = "0.32.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7753b721174eb8ff87a9a0e799e2d7bc3749323e773db92e0984debb00019d6e" +dependencies = [ + "bitflags", + "fallible-iterator", + "fallible-streaming-iterator", + "hashlink", + "libsqlite3-sys", + "smallvec", +] + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.5", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "shlex" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" + +[[package]] +name = "slab" +version = "0.4.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" + +[[package]] +name = "smallvec" +version = "1.15.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" + +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "12df2e0110f65b775f769bb17ef989067a1d931b2eb822bd4346631eeada89f9" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "unicode-ident" +version = "1.0.24" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" + +[[package]] +name = "vcpkg" +version = "0.2.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + +[[package]] +name = "waker-fn" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "317211a0dc0ceedd78fb2ca9a44aed3d7b9b26f81870d485c07122b4350673b7" + +[[package]] +name = "wasip2" +version = "1.0.4+wasi-0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" +dependencies = [ + "wit-bindgen", +] + +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" +dependencies = [ + "bitflags", +] + +[[package]] +name = "wstd" +version = "0.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29b52936db10a79bb724dadd1c2c3aac958e8229dcb1f1c7f2b7044ca9fc6a3a" +dependencies = [ + "anyhow", + "async-task", + "bytes", + "futures-lite", + "http", + "http-body", + "http-body-util", + "itoa", + "pin-project-lite", + "serde", + "serde_json", + "slab", + "wasip2", + "wstd-macro", +] + +[[package]] +name = "wstd-macro" +version = "0.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "153db9b65508bf6c2efe26617169840c03671ba5719a35868db7682a5261f7d4" +dependencies = [ + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zerocopy" +version = "0.8.56" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "556764e583adb45a9f8d413c2a147fa7e8d821e48e12b14fd560b607998b75eb" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.56" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f2ab42fc20575779bd240faa45f94a74256f755c0fa9e89f0ede20d91d0cdfc1" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/examples/guest-sqlite/Cargo.toml b/examples/guest-sqlite/Cargo.toml new file mode 100644 index 0000000..7730953 --- /dev/null +++ b/examples/guest-sqlite/Cargo.toml @@ -0,0 +1,26 @@ +[workspace] + +[package] +name = "guest-sqlite" +version = "0.1.0" +edition = "2021" +publish = false +description = "SQLite conformance and benchmark workload over wasi:filesystem" + +[[bin]] +name = "guest-sqlite" +path = "src/main.rs" + +[dependencies] +# Bundled SQLite: a real C database driving the filesystem through WASI, which +# is a far harsher test than any hand-written workload. Building it needs +# wasi-sdk — see the README. +rusqlite = { version = "0.32", features = ["bundled", "blob"] } +# Supplies the `http_server` attribute that makes this a component exporting +# `wasi:http/incoming-handler` — the only world the runtime serves. +wstd = "0.6" + +[profile.release] +opt-level = "s" +lto = true +codegen-units = 1 diff --git a/examples/guest-sqlite/src/main.rs b/examples/guest-sqlite/src/main.rs new file mode 100644 index 0000000..ad88016 --- /dev/null +++ b/examples/guest-sqlite/src/main.rs @@ -0,0 +1,1196 @@ +//! SQLite conformance and benchmark workload over `wasi:filesystem`. +//! +//! A real C database driving the filesystem is a far harsher test than any +//! hand-written workload. SQLite does small random reads and writes at page +//! granularity, rewrites a rollback journal on every transaction, truncates +//! and deletes it on commit, calls `fsync` at points where it genuinely needs +//! durability, and then — crucially — will tell us whether the bytes came back +//! correct via `PRAGMA integrity_check`. +//! +//! Every phase is timed, so this doubles as the benchmark. The numbers are +//! dominated by commit round trips rather than by SQLite: each `fsync` becomes +//! a transaction group, which is slab PUTs plus a signed root record. +//! +//! ## Journal mode +//! +//! `DELETE` — the default — is used deliberately. It creates, extends, +//! truncates and unlinks a `-journal` sidecar next to the database, which +//! exercises the parts of the filesystem a memory journal would skip entirely. +//! WAL is not usable: it needs shared memory that WASI does not provide. +//! +//! `locking_mode=EXCLUSIVE` because there is no `fcntl` locking under WASI, +//! and nothing else has the database open anyway. +//! +//! Answers `OK` when every phase passes and `integrity_check` returns `ok`, +//! and `FAIL: …` otherwise. +//! +//! ## Why this is an HTTP guest +//! +//! The workload below is unchanged — it is still a `wasi:cli`-shaped batch of +//! phases — but the runtime serves `wasi:http/proxy` and nothing else, so the +//! way to ask for it is a request rather than a process. `GET /` runs it once +//! and reports the verdict; the timing table goes to stdout, which the runtime +//! frames into its own log records rather than a terminal. +//! +//! One consequence worth knowing: the workload is not idempotent. It builds +//! its tables, so a second request against the same filesystem starts from +//! what the first left. That is the point when checking durability across +//! restarts, and a surprise if you expect a fresh database per request. + +use rusqlite::{params, Connection, OptionalExtension}; +use std::time::Instant; +use wstd::http::{Body, Error, Request, Response, StatusCode}; + +const DB_PATH: &str = "/sqlite/bench.db"; + +#[wstd::http_server] +async fn main(_req: Request) -> Result, Error> { + // The verdict is the body; the phase timings are stdout, and reach an + // operator through the runtime's guest log pipeline. + let (status, verdict) = match run() { + Ok(()) => (StatusCode::OK, "OK\n".to_string()), + Err(e) => { + println!("FAIL: {e}"); + (StatusCode::INTERNAL_SERVER_ERROR, format!("FAIL: {e}\n")) + } + }; + Ok(Response::builder() + .status(status) + .header("content-type", "text/plain; charset=utf-8") + .body(verdict.into()) + .expect("response is well formed")) +} + +/// Time a phase and report it in the benchmark table. +fn phase(name: &str, units: Option<(&str, u64)>, f: impl FnOnce() -> T) -> T { + let started = Instant::now(); + let out = f(); + let elapsed = started.elapsed(); + match units { + Some((unit, n)) if n > 0 && elapsed.as_secs_f64() > 0.0 => { + let rate = n as f64 / elapsed.as_secs_f64(); + println!(" {name:<28} {:>9.1} ms {rate:>10.0} {unit}/s", elapsed.as_secs_f64() * 1000.0); + } + _ => println!(" {name:<28} {:>9.1} ms", elapsed.as_secs_f64() * 1000.0), + } + out +} + +fn run() -> Result<(), String> { + let scale: usize = std::env::var("SQLITE_SCALE") + .ok() + .and_then(|s| s.parse().ok()) + .unwrap_or(2_000); + + if std::fs::metadata("/sqlite").is_err() { + std::fs::create_dir("/sqlite").map_err(|e| format!("create_dir /sqlite: {e}"))?; + } + + // Start from nothing so a rerun measures the same work. + for suffix in ["", "-journal", "-wal", "-shm"] { + let _ = std::fs::remove_file(format!("{DB_PATH}{suffix}")); + } + + println!("scale={scale} rows"); + println!("phase elapsed rate"); + + let conn = open()?; + schema(&conn)?; + bulk_insert(&conn, scale)?; + point_queries(&conn, scale)?; + updates_and_deletes(&conn, scale)?; + transactions_and_savepoints(&conn)?; + constraints(&conn)?; + joins_and_aggregates(&conn)?; + ctes_and_windows(&conn)?; + blobs(&conn)?; + incremental_blob_io(&conn)?; + text_and_collation(&conn)?; + triggers_and_views(&conn)?; + alter_and_indexes(&conn)?; + attach_database(&conn)?; + advanced_tables(&conn)?; + insert_variants(&conn)?; + transaction_modes(&conn)?; + datetime_and_scalars(&conn)?; + temp_tables(&conn)?; + commit_cost(&conn)?; + extensions(&conn)?; + // Drops come last so VACUUM has freed pages to reclaim, which is the case + // that actually shrinks the file. + drops(&conn)?; + maintenance(&conn)?; + integrity(&conn)?; + + drop(conn); + + // Reopen from scratch: everything above has to have reached the store, + // not merely SQLite's page cache. + let reopened = phase("reopen and verify", None, || -> Result<(), String> { + let conn = open()?; + let rows: i64 = conn + .query_row("SELECT count(*) FROM accounts", [], |r| r.get(0)) + .map_err(|e| format!("count after reopen: {e}"))?; + if rows == 0 { + return Err("database is empty after reopen".to_string()); + } + integrity(&conn) + }); + reopened?; + + Ok(()) +} + +fn open() -> Result { + let conn = Connection::open(DB_PATH).map_err(|e| format!("open {DB_PATH}: {e}"))?; + // No fcntl locking under WASI, and nothing else has the file open. + conn.pragma_update(None, "locking_mode", "EXCLUSIVE") + .map_err(|e| format!("locking_mode: {e}"))?; + // A real rollback journal, so the sidecar file's whole lifecycle is under + // test rather than being kept in memory. + conn.pragma_update(None, "journal_mode", "DELETE") + .map_err(|e| format!("journal_mode: {e}"))?; + // Every commit must actually reach the store; that is the point. + conn.pragma_update(None, "synchronous", "FULL") + .map_err(|e| format!("synchronous: {e}"))?; + conn.pragma_update(None, "foreign_keys", "ON") + .map_err(|e| format!("foreign_keys: {e}"))?; + // Temporary databases must live in memory, not on disk. + // + // SQLite finds a temp directory by probing candidates with `access(2)`, + // and WASI has no such call — wasi-libc's `faccessat` cannot answer, so + // every candidate is rejected and the search fails. VACUUM, which copies + // the database into a temporary one, then reports a bare "disk I/O error" + // with nothing to indicate that a missing syscall is the cause. + // + // This is a property of SQLite on WASI generally, not of this filesystem. + // Any guest doing more than trivial queries needs this line. + conn.pragma_update(None, "temp_store", "MEMORY") + .map_err(|e| format!("temp_store: {e}"))?; + Ok(conn) +} + +fn schema(conn: &Connection) -> Result<(), String> { + phase("schema (DDL)", None, || { + conn.execute_batch( + r#" + CREATE TABLE accounts ( + id INTEGER PRIMARY KEY, + name TEXT NOT NULL UNIQUE, + balance REAL NOT NULL DEFAULT 0.0 CHECK (balance >= 0), + kind TEXT NOT NULL DEFAULT 'std', + data BLOB, + created INTEGER NOT NULL + ); + + CREATE TABLE entries ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + account_id INTEGER NOT NULL REFERENCES accounts(id) ON DELETE CASCADE, + amount REAL NOT NULL, + memo TEXT, + UNIQUE (account_id, id) + ); + + CREATE TABLE audit ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + action TEXT NOT NULL, + subject INTEGER NOT NULL + ); + + CREATE INDEX idx_entries_account ON entries(account_id); + CREATE INDEX idx_accounts_kind ON accounts(kind, balance); + "#, + ) + .map_err(|e| format!("schema: {e}")) + }) +} + +fn bulk_insert(conn: &Connection, scale: usize) -> Result<(), String> { + phase("bulk insert", Some(("rows", scale as u64)), || { + conn.execute_batch("BEGIN").map_err(|e| e.to_string())?; + { + let mut stmt = conn + .prepare("INSERT INTO accounts (id, name, balance, kind, created) VALUES (?1, ?2, ?3, ?4, ?5)") + .map_err(|e| format!("prepare insert: {e}"))?; + for i in 0..scale { + let kind = if i % 3 == 0 { "premium" } else { "std" }; + stmt.execute(params![ + i as i64, + format!("account-{i:07}"), + (i as f64) * 1.5, + kind, + 1_700_000_000i64 + i as i64 + ]) + .map_err(|e| format!("insert {i}: {e}"))?; + } + } + conn.execute_batch("COMMIT").map_err(|e| e.to_string())?; + Ok::<_, String>(()) + })?; + + // Child rows, and a second commit so the journal lifecycle runs twice. + phase("bulk insert (children)", Some(("rows", (scale * 2) as u64)), || { + conn.execute_batch("BEGIN").map_err(|e| e.to_string())?; + { + let mut stmt = conn + .prepare("INSERT INTO entries (account_id, amount, memo) VALUES (?1, ?2, ?3)") + .map_err(|e| format!("prepare child insert: {e}"))?; + for i in 0..scale * 2 { + stmt.execute(params![ + (i % scale) as i64, + (i as f64) * 0.25, + if i % 7 == 0 { None } else { Some(format!("memo {i}")) } + ]) + .map_err(|e| format!("child insert {i}: {e}"))?; + } + } + conn.execute_batch("COMMIT").map_err(|e| e.to_string())?; + Ok::<_, String>(()) + })?; + + let n: i64 = conn + .query_row("SELECT count(*) FROM accounts", [], |r| r.get(0)) + .map_err(|e| e.to_string())?; + if n != scale as i64 { + return Err(format!("expected {scale} accounts, found {n}")); + } + Ok(()) +} + +fn point_queries(conn: &Connection, scale: usize) -> Result<(), String> { + let probes = scale.min(1_000); + phase("point select (by rowid)", Some(("queries", probes as u64)), || { + let mut stmt = conn + .prepare("SELECT name, balance FROM accounts WHERE id = ?1") + .map_err(|e| e.to_string())?; + for i in 0..probes { + let id = (i * 7919) % scale; + let (name, balance): (String, f64) = stmt + .query_row([id as i64], |r| Ok((r.get(0)?, r.get(1)?))) + .map_err(|e| format!("point select {id}: {e}"))?; + if name != format!("account-{id:07}") { + return Err(format!("wrong row for id {id}: {name}")); + } + if (balance - id as f64 * 1.5).abs() > 1e-9 { + return Err(format!("wrong balance for id {id}: {balance}")); + } + } + Ok::<_, String>(()) + })?; + + phase("indexed select (by name)", Some(("queries", probes as u64)), || { + let mut stmt = conn + .prepare("SELECT id FROM accounts WHERE name = ?1") + .map_err(|e| e.to_string())?; + for i in 0..probes { + let id = (i * 104_729) % scale; + let got: i64 = stmt + .query_row([format!("account-{id:07}")], |r| r.get(0)) + .map_err(|e| format!("select by name {id}: {e}"))?; + if got != id as i64 { + return Err(format!("index lookup returned {got} for {id}")); + } + } + Ok::<_, String>(()) + }) +} + +fn updates_and_deletes(conn: &Connection, scale: usize) -> Result<(), String> { + let n = scale.min(500); + phase("update", Some(("rows", n as u64)), || { + conn.execute_batch("BEGIN").map_err(|e| e.to_string())?; + let mut stmt = conn + .prepare("UPDATE accounts SET balance = balance + ?1 WHERE id = ?2") + .map_err(|e| e.to_string())?; + for i in 0..n { + stmt.execute(params![10.0f64, i as i64]) + .map_err(|e| format!("update {i}: {e}"))?; + } + drop(stmt); + conn.execute_batch("COMMIT").map_err(|e| e.to_string()) + })?; + + let balance: f64 = conn + .query_row("SELECT balance FROM accounts WHERE id = 0", [], |r| r.get(0)) + .map_err(|e| e.to_string())?; + if (balance - 10.0).abs() > 1e-9 { + return Err(format!("update did not apply: balance {balance}")); + } + + // UPSERT. + conn.execute( + "INSERT INTO accounts (id, name, balance, created) VALUES (?1, ?2, ?3, ?4) + ON CONFLICT(id) DO UPDATE SET balance = excluded.balance, kind = 'upserted'", + params![0i64, "account-0000000", 99.5f64, 1i64], + ) + .map_err(|e| format!("upsert: {e}"))?; + let (balance, kind): (f64, String) = conn + .query_row("SELECT balance, kind FROM accounts WHERE id = 0", [], |r| { + Ok((r.get(0)?, r.get(1)?)) + }) + .map_err(|e| e.to_string())?; + if (balance - 99.5).abs() > 1e-9 || kind != "upserted" { + return Err(format!("upsert wrong: {balance} {kind}")); + } + + // Cascading delete: removing an account must take its entries with it. + let victim = (scale - 1) as i64; + let before: i64 = conn + .query_row( + "SELECT count(*) FROM entries WHERE account_id = ?1", + [victim], + |r| r.get(0), + ) + .map_err(|e| e.to_string())?; + if before == 0 { + return Err("test setup: victim account has no entries".to_string()); + } + phase("cascading delete", None, || { + conn.execute("DELETE FROM accounts WHERE id = ?1", [victim]) + .map_err(|e| format!("delete: {e}")) + })?; + let after: i64 = conn + .query_row( + "SELECT count(*) FROM entries WHERE account_id = ?1", + [victim], + |r| r.get(0), + ) + .map_err(|e| e.to_string())?; + if after != 0 { + return Err(format!("foreign key cascade left {after} orphan entries")); + } + Ok(()) +} + +fn transactions_and_savepoints(conn: &Connection) -> Result<(), String> { + phase("rollback + savepoints", None, || { + // A plain rollback must leave nothing behind. This is where the + // journal file earns its keep: SQLite restores pages from it. + conn.execute_batch("BEGIN") + .and_then(|_| { + conn.execute_batch( + "INSERT INTO accounts (id, name, balance, created) + VALUES (900001, 'rolled-back', 1.0, 0)", + ) + }) + .and_then(|_| conn.execute_batch("ROLLBACK")) + .map_err(|e| format!("rollback: {e}"))?; + + let ghost: Option = conn + .query_row("SELECT id FROM accounts WHERE id = 900001", [], |r| r.get(0)) + .optional() + .map_err(|e| e.to_string())?; + if ghost.is_some() { + return Err("rolled-back row is still present".to_string()); + } + + // Nested savepoints: inner rolled back, outer kept. + conn.execute_batch( + "BEGIN; + INSERT INTO accounts (id, name, balance, created) VALUES (900002, 'keep', 1.0, 0); + SAVEPOINT inner; + INSERT INTO accounts (id, name, balance, created) VALUES (900003, 'drop', 1.0, 0); + ROLLBACK TO inner; + RELEASE inner; + COMMIT;", + ) + .map_err(|e| format!("savepoints: {e}"))?; + + let kept: i64 = conn + .query_row("SELECT count(*) FROM accounts WHERE id IN (900002, 900003)", [], |r| r.get(0)) + .map_err(|e| e.to_string())?; + if kept != 1 { + return Err(format!("savepoint semantics wrong: {kept} of 2 rows survived")); + } + Ok::<_, String>(()) + }) +} + +fn constraints(conn: &Connection) -> Result<(), String> { + phase("constraint enforcement", None, || { + // Each of these must be refused. A filesystem that lost writes could + // easily make a UNIQUE index stale, and this is how that shows up. + let cases: [(&str, &str); 5] = [ + ("UNIQUE", "INSERT INTO accounts (id, name, balance, created) VALUES (999001, 'account-0000000', 1.0, 0)"), + ("PRIMARY KEY", "INSERT INTO accounts (id, name, balance, created) VALUES (0, 'unique-name-a', 1.0, 0)"), + ("NOT NULL", "INSERT INTO accounts (id, name, balance, created) VALUES (999002, NULL, 1.0, 0)"), + ("CHECK", "INSERT INTO accounts (id, name, balance, created) VALUES (999003, 'unique-name-b', -5.0, 0)"), + ("FOREIGN KEY", "INSERT INTO entries (account_id, amount) VALUES (12345678, 1.0)"), + ]; + for (label, sql) in cases { + if conn.execute(sql, []).is_ok() { + return Err(format!("{label} constraint was not enforced")); + } + } + Ok::<_, String>(()) + }) +} + +fn joins_and_aggregates(conn: &Connection) -> Result<(), String> { + phase("joins + aggregates", None, || { + // Cross-check the same number two ways: a join with GROUP BY, and a + // correlated subquery. Disagreement means the index and the table + // disagree, which is exactly what a lossy filesystem produces. + let via_join: i64 = conn + .query_row( + "SELECT count(*) FROM ( + SELECT a.id FROM accounts a + JOIN entries e ON e.account_id = a.id + WHERE a.kind = 'premium' + GROUP BY a.id HAVING count(e.id) > 0)", + [], + |r| r.get(0), + ) + .map_err(|e| format!("join: {e}"))?; + let via_subquery: i64 = conn + .query_row( + "SELECT count(*) FROM accounts a WHERE a.kind = 'premium' + AND EXISTS (SELECT 1 FROM entries e WHERE e.account_id = a.id)", + [], + |r| r.get(0), + ) + .map_err(|e| format!("subquery: {e}"))?; + if via_join != via_subquery { + return Err(format!("join {via_join} != subquery {via_subquery}")); + } + + // LEFT JOIN must produce NULLs for accounts with no entries. + let _: i64 = conn + .query_row( + "SELECT count(*) FROM accounts a + LEFT JOIN entries e ON e.account_id = a.id WHERE e.id IS NULL", + [], + |r| r.get(0), + ) + .map_err(|e| format!("left join: {e}"))?; + + // Aggregate family. + let (total, avg, lo, hi): (f64, f64, f64, f64) = conn + .query_row( + "SELECT sum(amount), avg(amount), min(amount), max(amount) FROM entries", + [], + |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)), + ) + .map_err(|e| format!("aggregates: {e}"))?; + if !(lo <= avg && avg <= hi) || total <= 0.0 { + return Err(format!("aggregates inconsistent: {total} {avg} {lo} {hi}")); + } + Ok::<_, String>(()) + }) +} + +fn ctes_and_windows(conn: &Connection) -> Result<(), String> { + phase("CTEs + window functions", None, || { + // Recursive CTE: a self-contained arithmetic check that does not + // depend on the data, so a wrong answer means SQLite itself is + // misbehaving rather than the data being wrong. + let sum: i64 = conn + .query_row( + "WITH RECURSIVE seq(n) AS (SELECT 1 UNION ALL SELECT n+1 FROM seq WHERE n < 100) + SELECT sum(n) FROM seq", + [], + |r| r.get(0), + ) + .map_err(|e| format!("recursive CTE: {e}"))?; + if sum != 5050 { + return Err(format!("recursive CTE gave {sum}, expected 5050")); + } + + // Window function over real data. + let ranked: i64 = conn + .query_row( + "WITH ranked AS ( + SELECT id, row_number() OVER (PARTITION BY kind ORDER BY balance DESC) AS rn + FROM accounts) + SELECT count(*) FROM ranked WHERE rn = 1", + [], + |r| r.get(0), + ) + .map_err(|e| format!("window function: {e}"))?; + if ranked < 1 { + return Err("window function returned no partitions".to_string()); + } + Ok::<_, String>(()) + }) +} + +fn blobs(conn: &Connection) -> Result<(), String> { + // Large values force SQLite onto overflow pages, which means long runs of + // sequential page writes — a different access pattern from everything + // above, and the one that stresses the record and indirect-block layers. + let sizes = [4 * 1024usize, 256 * 1024, 2 * 1024 * 1024]; + let total: u64 = sizes.iter().map(|s| *s as u64).sum(); + + phase("blob write", Some(("bytes", total)), || { + conn.execute_batch("BEGIN").map_err(|e| e.to_string())?; + for (i, size) in sizes.iter().enumerate() { + let payload: Vec = (0..*size).map(|b| (b % 251) as u8).collect(); + conn.execute( + "INSERT INTO accounts (id, name, balance, created, data) VALUES (?1, ?2, 0, 0, ?3)", + params![800_000i64 + i as i64, format!("blob-{i}"), payload], + ) + .map_err(|e| format!("blob insert {i}: {e}"))?; + } + conn.execute_batch("COMMIT").map_err(|e| e.to_string()) + })?; + + phase("blob read + verify", Some(("bytes", total)), || { + for (i, size) in sizes.iter().enumerate() { + let got: Vec = conn + .query_row( + "SELECT data FROM accounts WHERE id = ?1", + [800_000i64 + i as i64], + |r| r.get(0), + ) + .map_err(|e| format!("blob read {i}: {e}"))?; + if got.len() != *size { + return Err(format!("blob {i}: expected {size} bytes, got {}", got.len())); + } + // Byte-for-byte, not just the length: this is the assertion that a + // corrupted or truncated block would fail. + if let Some(bad) = (0..*size).find(|b| got[*b] != (b % 251) as u8) { + return Err(format!("blob {i} corrupt at offset {bad}")); + } + } + Ok::<_, String>(()) + }) +} + +fn text_and_collation(conn: &Connection) -> Result<(), String> { + phase("text, types, collation", None, || { + conn.execute_batch( + "CREATE TABLE textish (id INTEGER PRIMARY KEY, s TEXT COLLATE NOCASE, r REAL, n INTEGER, x);", + ) + .map_err(|e| format!("create textish: {e}"))?; + + conn.execute( + "INSERT INTO textish (id, s, r, n, x) VALUES (1, ?1, ?2, ?3, NULL)", + params!["Grüße, Wörld — 日本語 🎉", 3.5f64, i64::MAX], + ) + .map_err(|e| format!("insert unicode: {e}"))?; + + let (s, r, n): (String, f64, i64) = conn + .query_row("SELECT s, r, n FROM textish WHERE id = 1", [], |row| { + Ok((row.get(0)?, row.get(1)?, row.get(2)?)) + }) + .map_err(|e| format!("read unicode: {e}"))?; + if s != "Grüße, Wörld — 日本語 🎉" { + return Err(format!("unicode round-trip failed: {s:?}")); + } + if r != 3.5 || n != i64::MAX { + return Err(format!("numeric round-trip failed: {r} {n}")); + } + + // NULL is distinct from everything, including itself. + let nulls: i64 = conn + .query_row("SELECT count(*) FROM textish WHERE x IS NULL", [], |r| r.get(0)) + .map_err(|e| e.to_string())?; + if nulls != 1 { + return Err("NULL handling wrong".to_string()); + } + + // COLLATE NOCASE on the column definition. + let ci: i64 = conn + .query_row("SELECT count(*) FROM textish WHERE s = ?1", params!["grüSSe, Wörld — 日本語 🎉"], |r| r.get(0)) + .map_err(|e| e.to_string())?; + // NOCASE is ASCII-only in SQLite, so this legitimately does not match; + // assert only that the query runs and returns a definite answer. + if ci > 1 { + return Err(format!("collation returned {ci} rows for a unique value")); + } + + // LIKE / GLOB / substr / replace. + let liked: i64 = conn + .query_row("SELECT count(*) FROM accounts WHERE name LIKE 'account-000000%'", [], |r| r.get(0)) + .map_err(|e| format!("LIKE: {e}"))?; + if liked == 0 { + return Err("LIKE matched nothing".to_string()); + } + Ok::<_, String>(()) + }) +} + +fn triggers_and_views(conn: &Connection) -> Result<(), String> { + phase("triggers + views", None, || { + conn.execute_batch( + r#" + CREATE VIEW premium_totals AS + SELECT a.id, a.name, count(e.id) AS entries, coalesce(sum(e.amount), 0) AS total + FROM accounts a LEFT JOIN entries e ON e.account_id = a.id + WHERE a.kind = 'premium' GROUP BY a.id; + + CREATE TRIGGER audit_delete AFTER DELETE ON accounts + BEGIN + INSERT INTO audit (action, subject) VALUES ('delete', OLD.id); + END; + "#, + ) + .map_err(|e| format!("create view/trigger: {e}"))?; + + let view_rows: i64 = conn + .query_row("SELECT count(*) FROM premium_totals", [], |r| r.get(0)) + .map_err(|e| format!("view query: {e}"))?; + if view_rows == 0 { + return Err("view returned no rows".to_string()); + } + + // Insert then delete, and check the trigger fired. + conn.execute( + "INSERT INTO accounts (id, name, balance, created) VALUES (700001, 'trigger-target', 1.0, 0)", + [], + ) + .map_err(|e| e.to_string())?; + conn.execute("DELETE FROM accounts WHERE id = 700001", []) + .map_err(|e| e.to_string())?; + + let audited: i64 = conn + .query_row("SELECT count(*) FROM audit WHERE subject = 700001", [], |r| r.get(0)) + .map_err(|e| e.to_string())?; + if audited != 1 { + return Err(format!("trigger did not fire: {audited} audit rows")); + } + Ok::<_, String>(()) + }) +} + +fn alter_and_indexes(conn: &Connection) -> Result<(), String> { + phase("ALTER TABLE + reindex", None, || { + conn.execute_batch( + "ALTER TABLE accounts ADD COLUMN nickname TEXT; + ALTER TABLE accounts RENAME COLUMN kind TO account_kind; + CREATE INDEX idx_accounts_nickname ON accounts(nickname);", + ) + .map_err(|e| format!("alter: {e}"))?; + + conn.execute("UPDATE accounts SET nickname = 'nick-' || id WHERE id < 50", []) + .map_err(|e| format!("populate nickname: {e}"))?; + + let found: i64 = conn + .query_row("SELECT id FROM accounts WHERE nickname = 'nick-7'", [], |r| r.get(0)) + .map_err(|e| format!("query renamed schema: {e}"))?; + if found != 7 { + return Err(format!("new index returned {found}")); + } + + conn.execute_batch("REINDEX;").map_err(|e| format!("reindex: {e}"))?; + + // The rename must be visible in the schema, not just tolerated. + let sql: String = conn + .query_row( + "SELECT sql FROM sqlite_master WHERE type='table' AND name='accounts'", + [], + |r| r.get(0), + ) + .map_err(|e| e.to_string())?; + if !sql.contains("account_kind") { + return Err("ALTER TABLE RENAME COLUMN not reflected in schema".to_string()); + } + Ok::<_, String>(()) + }) +} + +fn maintenance(conn: &Connection) -> Result<(), String> { + // VACUUM rewrites the entire database into a new file and swaps it in — + // the single heaviest filesystem operation SQLite performs. + phase("VACUUM", None, || { + conn.execute_batch("VACUUM;").map_err(|e| format!("vacuum: {e}")) + })?; + phase("ANALYZE", None, || { + conn.execute_batch("ANALYZE;").map_err(|e| format!("analyze: {e}")) + }) +} + +fn integrity(conn: &Connection) -> Result<(), String> { + phase("PRAGMA integrity_check", None, || { + let result: String = conn + .query_row("PRAGMA integrity_check", [], |r| r.get(0)) + .map_err(|e| format!("integrity_check: {e}"))?; + if result != "ok" { + return Err(format!("integrity_check: {result}")); + } + let fk: Option = conn + .query_row("PRAGMA foreign_key_check", [], |r| r.get(0)) + .optional() + .map_err(|e| format!("foreign_key_check: {e}"))?; + if let Some(violation) = fk { + return Err(format!("foreign_key_check found a violation in {violation}")); + } + Ok::<_, String>(()) + }) +} + +/// Incremental blob I/O: `sqlite3_blob_open` reads and writes *within* an +/// existing blob without materialising it. That is a partial-page update +/// pattern nothing else in this workload produces. +fn incremental_blob_io(conn: &Connection) -> Result<(), String> { + use std::io::{Read, Seek, SeekFrom, Write}; + + const SIZE: usize = 512 * 1024; + phase("incremental blob I/O", Some(("bytes", SIZE as u64)), || { + conn.execute_batch("CREATE TABLE incremental (id INTEGER PRIMARY KEY, payload BLOB);") + .map_err(|e| format!("create: {e}"))?; + // Reserve the space, then fill it in place. + conn.execute( + "INSERT INTO incremental (id, payload) VALUES (1, zeroblob(?1))", + [SIZE as i64], + ) + .map_err(|e| format!("zeroblob: {e}"))?; + + let mut blob = conn + .blob_open(rusqlite::DatabaseName::Main, "incremental", "payload", 1, false) + .map_err(|e| format!("blob_open: {e}"))?; + if blob.len() != SIZE { + return Err(format!("blob is {} bytes, expected {SIZE}", blob.len())); + } + + // Scattered writes rather than one sweep, so pages are dirtied out of + // order the way a real workload would. + for chunk in (0..SIZE / 4096).rev() { + let offset = chunk * 4096; + let bytes: Vec = (0..4096).map(|b| ((offset + b) % 251) as u8).collect(); + blob.seek(SeekFrom::Start(offset as u64)) + .map_err(|e| format!("seek {offset}: {e}"))?; + blob.write_all(&bytes) + .map_err(|e| format!("write {offset}: {e}"))?; + } + drop(blob); + + let mut blob = conn + .blob_open(rusqlite::DatabaseName::Main, "incremental", "payload", 1, true) + .map_err(|e| format!("blob_open read: {e}"))?; + let mut got = vec![0u8; SIZE]; + blob.read_exact(&mut got).map_err(|e| format!("read: {e}"))?; + if let Some(bad) = (0..SIZE).find(|b| got[*b] != (b % 251) as u8) { + return Err(format!("incremental blob corrupt at offset {bad}")); + } + Ok::<_, String>(()) + }) +} + +/// A second database file, open at the same time as the first. Two journals, +/// two sets of page writes, and a query spanning both. +fn attach_database(conn: &Connection) -> Result<(), String> { + phase("ATTACH + cross-database join", None, || { + for suffix in ["", "-journal"] { + let _ = std::fs::remove_file(format!("/sqlite/aux.db{suffix}")); + } + conn.execute_batch("ATTACH DATABASE '/sqlite/aux.db' AS aux;") + .map_err(|e| format!("attach: {e}"))?; + + conn.execute_batch( + "CREATE TABLE aux.labels (account_id INTEGER PRIMARY KEY, label TEXT NOT NULL); + INSERT INTO aux.labels (account_id, label) + SELECT id, 'label-' || id FROM main.accounts WHERE id < 100;", + ) + .map_err(|e| format!("populate attached: {e}"))?; + + // A join across the two files: both must be readable in one statement. + let joined: i64 = conn + .query_row( + "SELECT count(*) FROM main.accounts a JOIN aux.labels l ON l.account_id = a.id", + [], + |r| r.get(0), + ) + .map_err(|e| format!("cross-database join: {e}"))?; + if joined == 0 { + return Err("cross-database join returned nothing".to_string()); + } + + // Each attached database has its own integrity. + let aux_ok: String = conn + .query_row("PRAGMA aux.integrity_check", [], |r| r.get(0)) + .map_err(|e| format!("aux integrity: {e}"))?; + if aux_ok != "ok" { + return Err(format!("attached database integrity: {aux_ok}")); + } + + conn.execute_batch("DETACH DATABASE aux;") + .map_err(|e| format!("detach: {e}"))?; + if std::fs::metadata("/sqlite/aux.db").is_err() { + return Err("attached database file is missing after detach".to_string()); + } + Ok::<_, String>(()) + }) +} + +/// Table shapes with different on-disk representations: no rowid, stored +/// generated columns, and indexes that are not a plain column list. +fn advanced_tables(conn: &Connection) -> Result<(), String> { + phase("WITHOUT ROWID + generated cols", None, || { + conn.execute_batch( + r#" + CREATE TABLE kv (k TEXT PRIMARY KEY, v TEXT NOT NULL) WITHOUT ROWID; + + CREATE TABLE computed ( + a INTEGER NOT NULL, + b INTEGER NOT NULL, + virt INTEGER GENERATED ALWAYS AS (a + b) VIRTUAL, + store INTEGER GENERATED ALWAYS AS (a * b) STORED + ); + + -- Neither of these is a plain column list, so both take a + -- different path through the index machinery. + CREATE INDEX idx_partial ON accounts(balance) WHERE balance > 100.0; + CREATE INDEX idx_expr ON accounts(lower(name)); + "#, + ) + .map_err(|e| format!("create: {e}"))?; + + conn.execute_batch("BEGIN").map_err(|e| e.to_string())?; + { + let mut stmt = conn + .prepare("INSERT INTO kv (k, v) VALUES (?1, ?2)") + .map_err(|e| e.to_string())?; + for i in 0..500 { + stmt.execute(params![format!("key-{i:05}"), format!("value-{i}")]) + .map_err(|e| format!("kv insert {i}: {e}"))?; + } + } + conn.execute_batch("COMMIT").map_err(|e| e.to_string())?; + let v: String = conn + .query_row("SELECT v FROM kv WHERE k = 'key-00042'", [], |r| r.get(0)) + .map_err(|e| format!("without-rowid lookup: {e}"))?; + if v != "value-42" { + return Err(format!("WITHOUT ROWID returned {v}")); + } + + conn.execute("INSERT INTO computed (a, b) VALUES (6, 7)", []) + .map_err(|e| format!("generated insert: {e}"))?; + let (virt, store): (i64, i64) = conn + .query_row("SELECT virt, store FROM computed", [], |r| { + Ok((r.get(0)?, r.get(1)?)) + }) + .map_err(|e| format!("generated read: {e}"))?; + if virt != 13 || store != 42 { + return Err(format!("generated columns wrong: {virt} {store}")); + } + + // The expression index has to actually be usable. + let found: i64 = conn + .query_row( + "SELECT count(*) FROM accounts WHERE lower(name) = 'account-0000005'", + [], + |r| r.get(0), + ) + .map_err(|e| format!("expression index query: {e}"))?; + if found != 1 { + return Err(format!("expression index returned {found} rows")); + } + Ok::<_, String>(()) + }) +} + +fn insert_variants(conn: &Connection) -> Result<(), String> { + phase("INSERT OR REPLACE / IGNORE", None, || { + conn.execute( + "INSERT OR IGNORE INTO accounts (id, name, balance, created) VALUES (0, 'ignored', 1.0, 0)", + [], + ) + .map_err(|e| format!("or ignore: {e}"))?; + let name: String = conn + .query_row("SELECT name FROM accounts WHERE id = 0", [], |r| r.get(0)) + .map_err(|e| e.to_string())?; + if name == "ignored" { + return Err("INSERT OR IGNORE overwrote an existing row".to_string()); + } + + conn.execute( + "INSERT OR REPLACE INTO accounts (id, name, balance, created) VALUES (0, 'replaced', 5.0, 0)", + [], + ) + .map_err(|e| format!("or replace: {e}"))?; + let name: String = conn + .query_row("SELECT name FROM accounts WHERE id = 0", [], |r| r.get(0)) + .map_err(|e| e.to_string())?; + if name != "replaced" { + return Err(format!("INSERT OR REPLACE left {name}")); + } + Ok::<_, String>(()) + }) +} + +fn transaction_modes(conn: &Connection) -> Result<(), String> { + phase("BEGIN IMMEDIATE / EXCLUSIVE", None, || { + // Both take the write lock up front rather than on first write, which + // is a different ordering of journal creation. + for mode in ["IMMEDIATE", "EXCLUSIVE"] { + conn.execute_batch(&format!( + "BEGIN {mode}; + INSERT INTO audit (action, subject) VALUES ('{mode}', 1); + COMMIT;" + )) + .map_err(|e| format!("BEGIN {mode}: {e}"))?; + } + let n: i64 = conn + .query_row( + "SELECT count(*) FROM audit WHERE action IN ('IMMEDIATE', 'EXCLUSIVE')", + [], + |r| r.get(0), + ) + .map_err(|e| e.to_string())?; + if n != 2 { + return Err(format!("expected 2 rows from explicit modes, got {n}")); + } + Ok::<_, String>(()) + }) +} + +fn datetime_and_scalars(conn: &Connection) -> Result<(), String> { + phase("date/time + scalar functions", None, || { + // `now` reaches the host clock through wasi:clocks, so this quietly + // checks that too. + let today: String = conn + .query_row("SELECT date('now')", [], |r| r.get(0)) + .map_err(|e| format!("date('now'): {e}"))?; + if today.len() != 10 || !today.starts_with("20") { + return Err(format!("date('now') returned {today:?}")); + } + + let converted: String = conn + .query_row("SELECT datetime(1700000000, 'unixepoch')", [], |r| r.get(0)) + .map_err(|e| format!("datetime: {e}"))?; + if !converted.starts_with("2023-11-14") { + return Err(format!("unixepoch conversion returned {converted}")); + } + + let (upper, sub, replaced, padded): (String, String, String, String) = conn + .query_row( + "SELECT upper('abc'), substr('abcdef', 2, 3), replace('a-b-c', '-', '+'), printf('%05d', 42)", + [], + |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)), + ) + .map_err(|e| format!("scalar functions: {e}"))?; + if (upper.as_str(), sub.as_str(), replaced.as_str(), padded.as_str()) + != ("ABC", "bcd", "a+b+c", "00042") + { + return Err(format!("scalars wrong: {upper} {sub} {replaced} {padded}")); + } + Ok::<_, String>(()) + }) +} + +fn temp_tables(conn: &Connection) -> Result<(), String> { + phase("TEMP tables", None, || { + // With temp_store=MEMORY these never touch the filesystem, which is + // the point — the check is that they still work. + conn.execute_batch( + "CREATE TEMP TABLE scratch AS SELECT id, balance FROM accounts WHERE id < 200; + CREATE INDEX temp.idx_scratch ON scratch(balance);", + ) + .map_err(|e| format!("temp table: {e}"))?; + let n: i64 = conn + .query_row("SELECT count(*) FROM scratch", [], |r| r.get(0)) + .map_err(|e| e.to_string())?; + if n == 0 { + return Err("temp table is empty".to_string()); + } + conn.execute_batch("DROP TABLE scratch;") + .map_err(|e| format!("drop temp: {e}")) + }) +} + +/// The cost of a commit, measured on purpose. +/// +/// This is the number that matters most for anyone sizing a workload. Every +/// statement outside an explicit transaction is its own commit, and a commit +/// here is a transaction group: slab PUTs plus a signed root record, published +/// with a conditional PUT. That is two or more round trips to object storage, +/// and no amount of local caching can avoid them — durability is the point. +/// +/// Inside a transaction the same statements share one commit, which is why the +/// bulk phases above run three orders of magnitude faster per row. +fn commit_cost(conn: &Connection) -> Result<(), String> { + const N: usize = 25; + + conn.execute_batch("CREATE TABLE commit_cost (id INTEGER PRIMARY KEY, v TEXT);") + .map_err(|e| format!("create: {e}"))?; + + phase(" 25 inserts, autocommit", Some(("rows", N as u64)), || { + for i in 0..N { + conn.execute( + "INSERT INTO commit_cost (id, v) VALUES (?1, ?2)", + params![i as i64, "x"], + ) + .map_err(|e| format!("autocommit insert {i}: {e}"))?; + } + Ok::<_, String>(()) + })?; + + phase(" 25 inserts, one txn", Some(("rows", N as u64)), || { + conn.execute_batch("BEGIN").map_err(|e| e.to_string())?; + for i in 0..N { + conn.execute( + "INSERT INTO commit_cost (id, v) VALUES (?1, ?2)", + params![(1_000 + i) as i64, "x"], + ) + .map_err(|e| format!("txn insert {i}: {e}"))?; + } + conn.execute_batch("COMMIT").map_err(|e| e.to_string()) + })?; + + let n: i64 = conn + .query_row("SELECT count(*) FROM commit_cost", [], |r| r.get(0)) + .map_err(|e| e.to_string())?; + if n != (N * 2) as i64 { + return Err(format!("expected {} rows, got {n}", N * 2)); + } + Ok(()) +} + +/// Optional modules. Whether these are compiled in depends on how the bundled +/// SQLite was configured, so absence is reported rather than failed — but +/// anything present is exercised, because FTS5 in particular creates several +/// shadow tables and is a heavy filesystem workload. +fn extensions(conn: &Connection) -> Result<(), String> { + phase("optional modules", None, || { + let mut available = Vec::new(); + + // JSON is built into SQLite from 3.38 onwards. + match conn.query_row("SELECT json_extract('{\"a\":[1,2,3]}', '$.a[1]')", [], |r| { + r.get::<_, i64>(0) + }) { + Ok(2) => { + available.push("JSON"); + conn.execute_batch( + "CREATE TABLE docs (id INTEGER PRIMARY KEY, doc TEXT); + INSERT INTO docs (id, doc) VALUES (1, '{\"name\":\"x\",\"tags\":[\"a\",\"b\"]}'); + CREATE INDEX idx_docs_name ON docs(json_extract(doc, '$.name'));", + ) + .map_err(|e| format!("json table: {e}"))?; + let name: String = conn + .query_row( + "SELECT json_extract(doc, '$.name') FROM docs WHERE id = 1", + [], + |r| r.get(0), + ) + .map_err(|e| format!("json query: {e}"))?; + if name != "x" { + return Err(format!("json_extract returned {name}")); + } + } + Ok(other) => return Err(format!("json_extract returned {other}, expected 2")), + Err(_) => {} + } + + if conn + .execute_batch("CREATE VIRTUAL TABLE ft USING fts5(body);") + .is_ok() + { + available.push("FTS5"); + conn.execute_batch("BEGIN").map_err(|e| e.to_string())?; + { + let mut stmt = conn + .prepare("INSERT INTO ft (body) VALUES (?1)") + .map_err(|e| e.to_string())?; + for i in 0..2_000 { + stmt.execute([format!("document {i} about storage and enclaves")]) + .map_err(|e| format!("fts insert {i}: {e}"))?; + } + } + conn.execute_batch("COMMIT").map_err(|e| e.to_string())?; + + let hits: i64 = conn + .query_row("SELECT count(*) FROM ft WHERE ft MATCH 'enclaves'", [], |r| r.get(0)) + .map_err(|e| format!("fts match: {e}"))?; + if hits != 2_000 { + return Err(format!("FTS5 matched {hits} of 2000")); + } + // Rebuilds every shadow table from the content table. + conn.execute_batch("INSERT INTO ft(ft) VALUES('rebuild');") + .map_err(|e| format!("fts rebuild: {e}"))?; + } + + if conn + .execute_batch("CREATE VIRTUAL TABLE geo USING rtree(id, minX, maxX, minY, maxY);") + .is_ok() + { + available.push("R-Tree"); + conn.execute_batch("BEGIN").map_err(|e| e.to_string())?; + { + let mut stmt = conn + .prepare("INSERT INTO geo VALUES (?1, ?2, ?3, ?4, ?5)") + .map_err(|e| e.to_string())?; + for i in 0..1_000i64 { + let x = i as f64; + stmt.execute(params![i, x, x + 1.0, x, x + 1.0]) + .map_err(|e| format!("rtree insert {i}: {e}"))?; + } + } + conn.execute_batch("COMMIT").map_err(|e| e.to_string())?; + + let inside: i64 = conn + .query_row( + "SELECT count(*) FROM geo WHERE minX >= 10 AND maxX <= 21", + [], + |r| r.get(0), + ) + .map_err(|e| format!("rtree query: {e}"))?; + if inside == 0 { + return Err("R-Tree range query matched nothing".to_string()); + } + } + + println!( + " modules present: {}", + if available.is_empty() { + "none".to_string() + } else { + available.join(", ") + } + ); + Ok::<_, String>(()) + }) +} + +/// Dropping is what frees pages, and freeing is what VACUUM then reclaims. +fn drops(conn: &Connection) -> Result<(), String> { + phase("DROP index/view/trigger/table", None, || { + conn.execute_batch( + "DROP INDEX idx_accounts_nickname; + DROP INDEX idx_expr; + DROP VIEW premium_totals; + DROP TRIGGER audit_delete; + DROP TABLE kv; + DROP TABLE computed; + DROP TABLE incremental;", + ) + .map_err(|e| format!("drop: {e}"))?; + + // Gone from the schema, not merely inaccessible. + for (kind, name) in [ + ("index", "idx_accounts_nickname"), + ("view", "premium_totals"), + ("trigger", "audit_delete"), + ("table", "kv"), + ("table", "incremental"), + ] { + let present: Option = conn + .query_row( + "SELECT name FROM sqlite_master WHERE type = ?1 AND name = ?2", + params![kind, name], + |r| r.get(0), + ) + .optional() + .map_err(|e| e.to_string())?; + if present.is_some() { + return Err(format!("{kind} {name} survived DROP")); + } + } + + // And the dropped trigger must no longer fire. + conn.execute( + "INSERT INTO accounts (id, name, balance, created) VALUES (700002, 'no-trigger', 1.0, 0)", + [], + ) + .map_err(|e| e.to_string())?; + conn.execute("DELETE FROM accounts WHERE id = 700002", []) + .map_err(|e| e.to_string())?; + let audited: i64 = conn + .query_row("SELECT count(*) FROM audit WHERE subject = 700002", [], |r| r.get(0)) + .map_err(|e| e.to_string())?; + if audited != 0 { + return Err("dropped trigger still fired".to_string()); + } + Ok::<_, String>(()) + }) +} diff --git a/flake.lock b/flake.lock new file mode 100644 index 0000000..9a6e8cd --- /dev/null +++ b/flake.lock @@ -0,0 +1,64 @@ +{ + "nodes": { + "crane": { + "locked": { + "lastModified": 1785782307, + "narHash": "sha256-MPaRdVkf6zZP5fCPxYCi8Dr4pZzgmXzg8T9nVEbp3Mw=", + "owner": "ipetkov", + "repo": "crane", + "rev": "2c71e194474d13de031d729b729c968ddbe3507f", + "type": "github" + }, + "original": { + "owner": "ipetkov", + "repo": "crane", + "type": "github" + } + }, + "nixpkgs": { + "locked": { + "lastModified": 1786247143, + "narHash": "sha256-8S3Kcxs7D4UtxJxSJZz0m14CGhuW0MxfrIwJxeGWGnQ=", + "owner": "NixOS", + "repo": "nixpkgs", + "rev": "279b4a8275f032c566576b3f181fa0f27197f588", + "type": "github" + }, + "original": { + "owner": "NixOS", + "ref": "nixos-unstable", + "repo": "nixpkgs", + "type": "github" + } + }, + "root": { + "inputs": { + "crane": "crane", + "nixpkgs": "nixpkgs", + "rust-overlay": "rust-overlay" + } + }, + "rust-overlay": { + "inputs": { + "nixpkgs": [ + "nixpkgs" + ] + }, + "locked": { + "lastModified": 1786420161, + "narHash": "sha256-PU4CY4GUCDFOoS67dGrc/fS6Y+d1cYZSh6nbgYqAAwE=", + "owner": "oxalica", + "repo": "rust-overlay", + "rev": "3a646bd5daa3af78a78a1cb32329cb72d18fcae0", + "type": "github" + }, + "original": { + "owner": "oxalica", + "repo": "rust-overlay", + "type": "github" + } + } + }, + "root": "root", + "version": 7 +} diff --git a/flake.nix b/flake.nix new file mode 100644 index 0000000..481b717 --- /dev/null +++ b/flake.nix @@ -0,0 +1,627 @@ +{ + description = "Reproducible AWS Nitro Enclave image (EIF) for the s3fs enclave runtime"; + + # Why this exists at all: PCR0 is a digest of the enclave image, and a KMS key + # policy pins it while a client checks it. That number is only worth anything + # if someone else can rebuild the image and get the same one. The Dockerfile + # this replaces ran `apt-get update` on `debian:bookworm-slim`, so it produced + # a different PCR0 every week and attested to nothing reproducible. + # + # Only the *enclave* is built here. The parent instance is built with Packer, + # because the parent is the party the enclave excludes — nothing it contains + # is attested, so reproducing it buys no security property. + + inputs = { + nixpkgs.url = "github:NixOS/nixpkgs/nixos-unstable"; + + # nixpkgs' rustPlatform ships std for the host only. The guest component + # targets wasm32-wasip2, so the toolchain has to come from somewhere that + # can add targets. + rust-overlay = { + url = "github:oxalica/rust-overlay"; + inputs.nixpkgs.follows = "nixpkgs"; + }; + + crane.url = "github:ipetkov/crane"; + }; + + outputs = { self, nixpkgs, rust-overlay, crane }: + let + # An EIF is x86_64 only. Nothing here is meant to build elsewhere, and + # pretending otherwise would just produce a confusing failure. + system = "x86_64-linux"; + + pkgs = import nixpkgs { + inherit system; + overlays = [ (import rust-overlay) ]; + }; + + inherit (pkgs) lib; + + # Honours rust-toolchain.toml, plus the wasm target for the guest. + rustToolchain = pkgs.rust-bin.stable.latest.default.override { + targets = [ "wasm32-wasip2" "x86_64-unknown-linux-musl" ]; + }; + craneLib = (crane.mkLib pkgs).overrideToolchain rustToolchain; + + # `cleanCargoSource` drops everything that is not Rust. Two things here + # are not Rust and are not optional: + # + # .pem the AWS Nitro root, embedded by `include_str!` + # .wit the vendored WASI interfaces, read from disk by `bindgen!` + # + # Losing either fails deep in the build naming something else entirely — + # the WIT one surfaces as `could not find 'wasi' in 'bindings'`. + srcFilter = path: type: + # Guests are separate workspaces. Including their sources here changes + # the runtime's store path (and its image's PCR0) on guest-only edits. + !(lib.hasPrefix "${toString ./.}/examples/" (toString path) + || toString path == "${toString ./.}/examples") + && ((lib.hasSuffix ".pem" path) + || (lib.hasSuffix ".wit" path) + || (craneLib.filterCargoSources path type)); + + workspaceSrc = lib.cleanSourceWith { + src = ./.; + filter = srcFilter; + name = "source"; + }; + + # aws-lc-sys compiles a bundled C library; blake3 and friends want a C + # compiler too. This is the usual place a Rust-in-Nix build stops. + nativeArgs = { + strictDeps = true; + nativeBuildInputs = with pkgs; [ cmake pkg-config perl ]; + # OpenSSL, for `openssl-sys` under `webauthn-rs`. It is here reluctantly + # and the reluctance is the point: this is a second crypto stack, with a + # native library, entering the closure that PCR0 measures — alongside + # aws-lc-rs, which already provides every primitive it is used for. + # + # It is also load-bearing rather than cosmetic. Without it the enclave + # image does not build at all, which is how the cost first showed up: + # a host cargo build succeeded against the system OpenSSL and the + # reproducible build did not. + buildInputs = [ pkgs.openssl ]; + # aws-lc-sys drives cmake itself; letting Nix's cmake hook configure + # the crate's own build tree makes it fail confusingly. + dontUseCmakeConfigure = true; + }; + + # ---- the runtime ----------------------------------------------------- + runtimeArgs = nativeArgs // { + pname = "enclave-runtime"; + version = "0.1.0"; + src = workspaceSrc; + # `--bin` so crane builds the binary rather than also the library's + # test targets. + cargoExtraArgs = "--locked -p enclave-runtime --bin enclave-runtime"; + # The workspace has integration tests needing a built wasm guest and a + # running MinIO. `nix flake check` runs the unit tests instead. + doCheck = false; + }; + + runtimeDeps = craneLib.buildDepsOnly runtimeArgs; + enclave-runtime = craneLib.buildPackage (runtimeArgs // { + cargoArtifacts = runtimeDeps; + }); + + # ---- the verifier ---------------------------------------------------- + # A client-side tool, built from the same tree so it cannot drift from + # the runtime it checks. + attestArgs = nativeArgs // { + pname = "nitro-attest"; + version = "0.1.0"; + src = workspaceSrc; + cargoExtraArgs = "--locked -p nitro-attestation --features cli"; + doCheck = false; + }; + + nitro-attest = craneLib.buildPackage (attestArgs // { + cargoArtifacts = craneLib.buildDepsOnly attestArgs; + }); + + # ---- the guest ------------------------------------------------------- + # A separate workspace with its own lock file, and a different target, so + # it cannot share the runtime's dependency layer. + guestArgs = { + pname = "guest-http"; + version = "0.1.0"; + src = lib.cleanSourceWith { + src = ./examples/guest-http; + filter = path: type: + lib.hasSuffix ".wit" path || craneLib.filterCargoSources path type; + name = "guest-source"; + }; + strictDeps = true; + CARGO_BUILD_TARGET = "wasm32-wasip2"; + cargoExtraArgs = "--locked"; + doCheck = false; + }; + + guestDeps = craneLib.buildDepsOnly guestArgs; + guest-http = craneLib.buildPackage (guestArgs // { + cargoArtifacts = guestDeps; + }); + + # What an operator uploads to the roots bucket, and the PCR16 a key policy + # pins for it. The guest is not in the enclave image: the runtime fetches + # `deployment.guestObject` at boot and measures it into PCR16. + # + # `guest-pcr16.json` comes from `nitro-attest --measure`, the function a + # client verifies with, so the number in the policy and the number a + # client checks cannot disagree. + guest-release = pkgs.runCommand "guest-release" + { nativeBuildInputs = [ nitro-attest pkgs.jq ]; } + '' + mkdir -p $out + cp ${guest-http}/bin/guest-http.wasm $out/guest.wasm + nitro-attest --measure $out/guest.wasm > $out/guest-pcr16.json + jq -e '.PCR16 | length == 96' $out/guest-pcr16.json > /dev/null \ + || { echo "nitro-attest did not report a PCR16" >&2; exit 1; } + echo "PCR16 $(jq -r .PCR16 $out/guest-pcr16.json)" + ''; + + # ---- the self-test payload ------------------------------------------- + # Static musl, so the entropy harness keeps its tiny ramdisk with no + # closure at all. It only uses rustix, anyhow and ciborium, which is why + # this one can be static where the runtime cannot. + selftestArgs = { + pname = "nsm-selftest"; + version = "0.1.0"; + src = workspaceSrc; + strictDeps = true; + CARGO_BUILD_TARGET = "x86_64-unknown-linux-musl"; + CARGO_BUILD_RUSTFLAGS = "-C target-feature=+crt-static"; + cargoExtraArgs = "--locked -p nitro-nsm --bin nsm-selftest"; + doCheck = false; + }; + + nsm-selftest = craneLib.buildPackage (selftestArgs // { + cargoArtifacts = craneLib.buildDepsOnly selftestArgs; + }); + + # ---- eif_build ------------------------------------------------------- + eif-build = pkgs.rustPlatform.buildRustPackage rec { + pname = "eif_build"; + version = "0.6.0"; + + src = pkgs.fetchFromGitHub { + owner = "aws"; + repo = "aws-nitro-enclaves-image-format"; + rev = "v${version}"; + hash = "sha256-d70XEPRY/dCgYJOCOImpOFuwNGcTxBj6FTA17Rp9l20="; + }; + + # Upstream gitignores its lock file, so one is kept here. That is not + # a workaround: eif_build is the tool that *computes PCR0*, and letting + # its dependency set float would mean the measurement was produced by + # something slightly different each time. + postPatch = "cp ${./deploy/nix/eif_build-Cargo.lock} Cargo.lock"; + cargoLock.lockFile = ./deploy/nix/eif_build-Cargo.lock; + cargoBuildFlags = [ "-p" "eif_build" ]; + doCheck = false; + + # openssl-sys, for the EIF's signing support. + nativeBuildInputs = [ pkgs.pkg-config ]; + buildInputs = [ pkgs.openssl ]; + + meta.description = "Assembles an Enclave Image Format file"; + }; + + # ---- AWS's enclave kernel and bootstrap ------------------------------ + # The kernel, its config, the cmdline, `init` and the NSM driver. Pinned + # by hash so every build uses identical bytes. + # + # These are prebuilt binaries whose provenance is taken on trust — the + # one opaque input to PCR0. AWS publishes aws-nitro-enclaves-sdk-bootstrap, + # which builds them from source with nixpkgs; moving to it would close + # that gap at the cost of a kernel compile. + blobs = pkgs.fetchFromGitHub { + owner = "aws"; + repo = "aws-nitro-enclaves-cli"; + rev = "v1.4.5"; + hash = "sha256-pdTbmsf7Kj7uF2g8zN6ur8gBJWOEkpb9YOLc2v0xnxQ="; + sparseCheckout = [ "blobs/x86_64" ]; + }; + + gvproxy-static = pkgs.gvproxy.overrideAttrs (old: { + env = (old.env or { }) // { CGO_ENABLED = "0"; }; + }); + + callEif = pkgs.callPackage ./nix/eif.nix { + inherit blobs eif-build; + }; + + # gvforwarder brings the tap device up and then shells out to a DHCP + # client for its address — `udhcpc` if present, otherwise `dhclient`. + # Neither is in the runtime's closure, and the symptom is a forwarder + # that connects to gvproxy, is disconnected immediately, and retries + # forever while the gateway never answers. The cause is only visible if + # the child's stderr is not discarded, which is why it now isn't. + # + # busybox supplies udhcpc and the applets its lease script needs, as one + # static binary. Hand-rolling netlink instead would save ~1 MB and cost a + # reimplementation of the part of gvforwarder that already works. + # + # It goes in as a whole store path rather than a copied binary, because + # udhcpc does not configure the interface itself — it execs a lease + # script, and nixpkgs patches busybox to look for that script *inside its + # own store path*. Copy out just `bin/busybox` and the client obtains a + # perfectly good lease and then applies none of it, silently. Shipping + # the closure makes the script, its `#!` line and the applets it calls + # all resolve, and needs no hand-written replacement. + busybox = pkgs.pkgsStatic.busybox; + + # Which store the production image belongs to. Baked in, so PCR0 covers + # it — see deploy/nix/deployment.nix for why that has to be true. + deployment = import ./deploy/nix/deployment.nix; + + # What both the production and emulator images are made of. Only the + # environment differs between them. + # The same runtime, built with `testing`, for the QEMU emulator only. + # + # It exists for one reason: the emulator has no domain and no CA it can + # reach, so it cannot obtain an ACME certificate — and the production + # binary refuses to serve anything else. `testing` restores + # `--tls self-signed` and, with it, `rcgen`. + # + # This is a **different binary with a different PCR0**, which is the point: + # nothing here can be mistaken for the production image, and the e2e + # asserts its PCR0 against its own build rather than a published one. The + # production image contains no certificate generator at all. + enclave-runtime-testing = craneLib.buildPackage (runtimeArgs // { + pname = "enclave-runtime-testing"; + cargoArtifacts = runtimeDeps; + cargoExtraArgs = + "--locked -p enclave-runtime --bin enclave-runtime --features testing"; + }); + + runtimeImage = { + name = "s3fs"; + # No guest here. It is fetched from the store at boot and measured + # into PCR16 — see `guest-release` and S3FS_GUEST_OBJECT below. + payload = { + "enclave-runtime" = "${enclave-runtime}/bin/enclave-runtime"; + "usr/local/bin/gvforwarder" = "${gvproxy-static}/bin/gvforwarder"; + }; + closureRoots = [ enclave-runtime busybox pkgs.cacert ]; + command = "/enclave-runtime"; + env = { + # Without a trust store the AWS SDK panics with "no CA certificates + # found" before it makes a single request — including against a + # plaintext endpoint, since the check happens when the client is + # built rather than when it connects. + # + # This is the enclave's *own* trust store, shipped inside the image + # and covered by PCR0. It is what makes "the parent cannot redirect + # us" true: the parent answers DNS, so the only thing stopping it + # pointing S3 at itself is that the certificate would not validate + # against these roots. + SSL_CERT_FILE = "${pkgs.cacert}/etc/ssl/certs/ca-bundle.crt"; + # gvforwarder finds its DHCP client with exec.LookPath, so busybox's + # own bin directory goes on PATH — no symlinks, and the lease script + # it execs resolves through the same closure. + PATH = "${busybox}/bin:/usr/local/bin"; + # Where the guest comes from — not the guest, which is not in the + # image. The runtime fetches this key from the roots bucket, extends + # PCR16 with the object's hash and locks the register before it asks + # KMS for a key. The location is measured here, by PCR0; what arrives + # is measured there, by PCR16, so the object need not be trusted. + S3FS_GUEST_OBJECT = deployment.guestObject; + S3FS_BACKGROUND_TASKS = lib.boolToString deployment.backgroundTasks; + S3FS_BACKGROUND_CONCURRENCY = toString deployment.backgroundConcurrency; + # Push notifications. Empty means off. The credential is named, never + # carried: baking a service-account key into the image would put it in + # every copy of the image and pin it to a PCR0 it has no reason to + # move with. + S3FS_FCM_PROJECT_ID = deployment.fcmProjectId; + S3FS_FCM_SERVICE_ACCOUNT_PARAMETER = deployment.fcmServiceAccountParameter; + S3FS_HTTP_LISTEN = "0.0.0.0:443"; + # ACME, not self-signed: a platform authenticator will not attest + # against a certificate a browser does not trust, so a self-signed one + # would mean no passkey could ever register. + S3FS_TLS = "acme"; + S3FS_NETWORK = "gvproxy"; + S3FS_GVFORWARDER = "/usr/local/bin/gvforwarder"; + S3FS_RANDOM_SOURCE = "nsm"; + S3FS_CLOCK_SOURCE = "ptp"; + + # The store this image is for. Attested, because a receipt only + # means "no filesystem here" if the host cannot choose "here". + S3FS_BUCKET = deployment.dataBucket; + S3FS_ROOTS_BUCKET = deployment.rootsBucket; + S3FS_BUCKET_PREFIX = deployment.bucketPrefix; + S3FS_ID = deployment.fsId; + AWS_REGION = deployment.region; + S3FS_TLS_DOMAINS = lib.concatStringsSep "," deployment.tlsDomains; + + # A production image demands a receipt signed by AWS. The emulator + # image overrides this, and because the environment is measured, + # PCR0 tells a client which kind it is talking to. + S3FS_RECEIPT_TRUST = "required"; + + # The master secret is minted by KMS inside the enclave and released + # only against an attestation whose PCR0 and PCR16 match the key + # policy. A wrong image or a wrong guest does not get a refused + # mount — it gets no key at all. + # + # S3FS_MASTER_KEY is deliberately absent, and the runtime *refuses* + # to start if it is set alongside this: a key from configuration is a + # key the parent instance holds, which is the whole thing this + # prevents. S3FS_KMS_KEY_ID and S3FS_MASTER_KEY_PARAMETER come from + # the deployment, since they name resources this repository does not + # own. + S3FS_MASTER_KEY_SOURCE = "kms"; + + # Guest stdout and stderr go to CloudWatch as well as the console. + # The enclave calls PutLogEvents itself, over the same path it uses + # for S3 and KMS, so TLS terminates inside and the parent carries + # ciphertext. The group and stream are created by Terraform; this + # runtime holds `logs:PutLogEvents` and cannot make them. + S3FS_GUEST_LOG_GROUP = deployment.guestLogGroup; + S3FS_GUEST_LOG_STREAM = deployment.guestLogStream; + + # Every request that could reach the guest needs a fresh WebAuthn + # assertion bound to exactly that request. Without an RP id the + # runtime serves the guest to anyone who can open a connection and + # says so at startup — which is a development arrangement, not this + # one. The domain must be the one the app's passkeys are scoped to, + # and it must be browser-trusted, hence ACME rather than self-signed. + S3FS_WEBAUTHN_RP_ID = deployment.rpId; + S3FS_WEBAUTHN_ORIGIN = "https://" + deployment.rpId; + S3FS_WEBAUTHN_ALLOWED_ORIGINS = lib.concatStringsSep "," deployment.webauthnAllowedOrigins; + # + # Registration is open: anyone who can reach the port may create a + # tenant of their own. There is nothing to provision, and nothing an + # image could leak by carrying it. + } + # Set only when used, so an image that names no origins and keeps the + # default timeout is the image it was before these settings existed. + // lib.optionalAttrs (deployment.guestEgressOrigins != [ ]) { + S3FS_GUEST_EGRESS_ORIGINS = lib.concatStringsSep "," deployment.guestEgressOrigins; + } + // lib.optionalAttrs (deployment.backgroundTimeoutSecs != null) { + S3FS_BACKGROUND_TIMEOUT_SECS = toString deployment.backgroundTimeoutSecs; + } + // deployment.guestEnv; + }; + + # The runtime configured for the emulator, as a function of the relying + # party so a client developer can boot one their app can sign for: + # + # nix build --impure --expr '(builtins.getFlake "git+file://$PWD").lib.x86_64-linux.eifQemu + # { rpId = "example.com"; allowedOrigins = [ "android:apk-key-hash:..." ]; }' + # + # which is what `dev-enclave.sh --rp-id --allowed-origin` does. The + # defaults are `packages.eif-qemu`, the image the e2e boots. + eifQemu = { + rpId ? "enclave.test", + allowedOrigins ? [ ], + # Origins the guest may reach — see `guestEgressOrigins` in + # deploy/nix/deployment.nix. The dev stack's host is 192.168.127.254. + guestEgressOrigins ? [ ], + backgroundTimeoutSecs ? null, + # Variables for the guest, e.g. the address of the service it may reach. The guest inherits + # the image environment minus `AWS_*` and `S3FS_*`, so a plain name reaches it. + guestEnv ? { }, + # The certificate. The defaults are Pebble on the harness host, for `enclave.test`. An emulator + # on a public host passes its real name and a real CA instead — `dev-enclave.sh --domain` — + # with `acmeCa = null`, since a public CA's API needs no trust root shipped in the image. + tlsDomains ? [ "enclave.test" ], + acmeDirectory ? "https://192.168.127.254:14000/dir", + acmeCa ? "/pebble-ca.pem", + acmeContacts ? [ ], + # Real notifications: `{ projectId; serviceAccountFile; }`, the file a Firebase service + # account's JSON key. Null keeps the stub. The credential is baked into the image like every + # other setting here, so it lands in the Nix store and the EIF of whoever builds this. + fcm ? null, + }: callEif (runtimeImage // { + payload = runtimeImage.payload // { + "enclave-runtime" = "${enclave-runtime-testing}/bin/enclave-runtime"; + # Pebble's ACME API is served under a certificate no public root + # signed, so its root ships *inside* the image and is covered by + # PCR0 like every other piece of configuration. A test root in a + # test image: the production EIF has no such file, and the + # production binary has no flag that would read one. + "pebble-ca.pem" = ./deploy/qemu-nitro/pebble/ca.pem; + }; + closureRoots = [ enclave-runtime-testing busybox pkgs.cacert ]; + name = "s3fs-qemu"; + env = runtimeImage.env // { + S3FS_CLOCK_SOURCE = "host"; + + # QEMU's NSM does not sign at all: its source says "we don't + # actually sign the data, so we use -1 as the 'alg' value", and -1 + # is not a COSE algorithm. A client meeting one of its documents has + # to skip the signature, the chain and the validity windows — most + # of what a client does, and precisely the part worth exercising + # before it meets hardware. + # + # So the runtime re-signs what the device produced, contents + # untouched, with a chain it mints at boot. It says nothing about + # *who* produced a document — the key is inside an image its + # operator controls — but it means the client code developed against + # the emulator is the code that runs against Nitro, rather than a + # relaxed variant of it. The root is reported on the console at + # startup; `deploy/qemu-nitro/lib.sh` captures it for the clients. + # + # Testing-only, like `S3FS_ACME_CA`: the production binary has no + # such flag, so no deployment can sign its own attestations. + S3FS_COSIGN_ATTESTATIONS = "1"; + + # And receipts stay content-checked, because that chain is minted + # fresh at every boot. A receipt signed at genesis names a root the + # next boot no longer has, so requiring a signature here would turn + # every restart into a refusal to resume. Contents are still + # verified — PCR0, PCR16 and the state_root — which is the whole + # boot machine minus the one part that needs real hardware. + S3FS_RECEIPT_TRUST = "unsigned-emulator"; + # A directory per client, which the e2e exercises with two client + # certificates. Concurrency has to rise with it or every client + # still queues behind every other and the per-client locks buy + # nothing. + # And it cannot use KMS at all, for the same reason: KMS verifies + # the attestation document carrying the enclave's recipient public + # key, and will not accept one that is unsigned. So the emulator + # keeps the development key source. PCR0 differs between the two + # images, so a client can tell which it is talking to. + S3FS_MASTER_KEY_SOURCE = "static"; + # Real ACME, against a Pebble running on the host — the same code + # path production takes, which is the whole reason to prefer it. + # + # This image used to serve a self-signed certificate, because there + # is no public domain here and nothing Let's Encrypt could reach, so + # an order failed forever and stalled every handshake rather than + # failing loudly. The consequence was worse than the workaround: the + # e2e proved a TLS mode a production build cannot even parse. Pebble + # removes the reason, so the mode went with it. + # + # `enclave.test` resolves, inside the Pebble container, to the host + # loopback where gvproxy forwards :443 into this enclave — so the + # TLS-ALPN-01 challenge arrives on the same port the service uses, + # which is exactly the arrangement production runs. + S3FS_TLS = "acme"; + S3FS_TLS_DOMAINS = nixpkgs.lib.concatStringsSep "," tlsDomains; + S3FS_ACME_DIRECTORY = acmeDirectory; + # The e2e schedules work and waits for it to run. Production leaves + # this off in deployment.nix; turning it on here changes only the + # emulator's PCR0, and the harness checks PCR0 against its own + # `nix build` rather than against a published number, so nothing + # downstream moves. The guest is guest-http, which exports the + # `run-task` the runtime refuses to start without when this is set. + S3FS_BACKGROUND_TASKS = "true"; + # Notifications, against a stub on the host rather than Google — + # which is unreachable from here and would refuse an invented + # registration token anyway. What the e2e checks is what the + # *runtime* sends, and the stub records exactly that. + # + # A literal credential, not an SSM parameter: there is no SSM here. + # It is a throwaway key in a test image, and the `http://` endpoint + # is the same downgrade `--guest-log-endpoint` already is. PCR0 + # records that this image was built with both. + S3FS_FCM_PROJECT_ID = "e2e"; + S3FS_FCM_SERVICE_ACCOUNT = builtins.readFile ./deploy/qemu-nitro/fcm/service-account.json; + # Off. Inherited from the production image, and there is no AWS + # here to send to: the harness's credentials are MinIO's, which + # CloudWatch would reject. Empty means off — the same shape as the + # TLS override above, and the reason `guest_log_config` treats an + # empty setting as unset rather than as a typo. + S3FS_GUEST_LOG_GROUP = ""; + S3FS_GUEST_LOG_STREAM = ""; + # The gate, with a relying party the harness can drive. The e2e + # exercises it with the software passkey the `testing` feature + # provides, so the default is a name nothing else answers to. + # + # A developer testing a phone app against this passes their own: a + # platform authenticator creates a passkey only for an rp id whose + # domain publishes assetlinks.json naming the app, and the app claims + # `android:apk-key-hash:` rather than the web origin. The rp id + # is independent of the certificate's name, which stays enclave.test. + S3FS_WEBAUTHN_RP_ID = rpId; + S3FS_WEBAUTHN_ORIGIN = "https://${rpId}"; + S3FS_ENDPOINT = "http://192.168.127.254:9000"; + S3FS_FORCE_PATH_STYLE = "1"; + S3FS_BUCKET = "e2e-data"; + S3FS_ROOTS_BUCKET = "e2e-roots"; + S3FS_MASTER_KEY = "00000000000000000000000000000000000000000000000000000000000000ab"; + AWS_ACCESS_KEY_ID = "minioadmin"; + AWS_SECRET_ACCESS_KEY = "minioadmin"; + AWS_REGION = "us-east-1"; + RUST_LOG = "info,s3fs=debug"; + } // nixpkgs.lib.optionalAttrs (acmeCa != null) { + S3FS_ACME_CA = acmeCa; + } // nixpkgs.lib.optionalAttrs (acmeContacts != [ ]) { + S3FS_ACME_CONTACTS = nixpkgs.lib.concatStringsSep "," acmeContacts; + } // nixpkgs.lib.optionalAttrs (fcm == null) { + S3FS_FCM_ENDPOINT = "http://192.168.127.254:9101"; + } // nixpkgs.lib.optionalAttrs (fcm != null) { + S3FS_FCM_PROJECT_ID = fcm.projectId; + # One line: the image's environment file is `KEY=value` per line. + S3FS_FCM_SERVICE_ACCOUNT = builtins.toJSON (builtins.fromJSON (builtins.readFile fcm.serviceAccountFile)); + } // nixpkgs.lib.optionalAttrs (allowedOrigins != [ ]) { + S3FS_WEBAUTHN_ALLOWED_ORIGINS = nixpkgs.lib.concatStringsSep "," allowedOrigins; + } // nixpkgs.lib.optionalAttrs (guestEgressOrigins != [ ]) { + S3FS_GUEST_EGRESS_ORIGINS = nixpkgs.lib.concatStringsSep "," guestEgressOrigins; + } // nixpkgs.lib.optionalAttrs (backgroundTimeoutSecs != null) { + S3FS_BACKGROUND_TIMEOUT_SECS = toString backgroundTimeoutSecs; + } // guestEnv; + }); + + in + { + lib.${system} = { inherit eifQemu; }; + + packages.${system} = { + inherit enclave-runtime guest-http guest-release nsm-selftest nitro-attest eif-build blobs; + + # Both ends of the vsock from one package: `make build` emits gvproxy + # for the parent and gvforwarder for the enclave, so they cannot drift. + # + # Built with CGO disabled, which upstream already does for gvforwarder + # but not for gvproxy. Nix's gvproxy is dynamically linked against a + # glibc in the store, so it runs nowhere that lacks that exact store + # path — not on this host outside a Nix namespace, and not on the + # Amazon Linux parent the AMI bakes. Static, it is one file that runs + # anywhere, which is what a binary crossing out of Nix needs to be. + gvproxy = gvproxy-static; + + # The production image. + eif = callEif runtimeImage; + + # The same runtime, configured for the emulator. Neither image contains + # the guest: the e2e uploads `guest-release` to MinIO, at the key + # `deployment.guestObject` names, and the enclave measures it there. + # + # A different image, and therefore a *different PCR0* — which is the + # honest outcome: an enclave image is its configuration as much as its + # code, and two configurations cannot share a measurement. The e2e + # proves the machinery, not that production's exact bytes booted. + # + # Two things have to differ. QEMU's nitro-enclave machine has no PTP + # device, so the trusted clock falls back to the system one. And the + # store is MinIO on the host, reachable at gvproxy's host address. + # The credentials here are test values in a test image; nothing that + # matters is protected by them. + eif-qemu = eifQemu { }; + + # The entropy harness's image: no closure, no network, nothing but the + # device check. + eif-selftest = callEif { + name = "selftest"; + payload = { "nsm-selftest" = "${nsm-selftest}/bin/nsm-selftest"; }; + command = "/nsm-selftest"; + env = { }; + withClosure = false; + }; + + default = self.packages.${system}.eif; + }; + + devShells.${system}.default = pkgs.mkShell { + inputsFrom = [ enclave-runtime ]; + packages = with pkgs; [ + rustToolchain + cmake + pkg-config + gvproxy + cpio + jq + eif-build + opentofu + packer + ]; + }; + + checks.${system} = { + inherit enclave-runtime guest-http nsm-selftest; + + workspace-tests = craneLib.cargoTest (runtimeArgs // { + cargoArtifacts = runtimeDeps; + cargoTestExtraArgs = "--workspace --lib"; + doCheck = true; + }); + }; + + formatter.${system} = pkgs.nixpkgs-fmt; + }; +} diff --git a/nix/eif.nix b/nix/eif.nix new file mode 100644 index 0000000..e5b7e1a --- /dev/null +++ b/nix/eif.nix @@ -0,0 +1,185 @@ +# Assemble an Enclave Image Format file. +# +# The layout is not a matter of taste — it is what AWS's `init` expects, and +# each detail below was learned by booting an image that failed in a way that +# named something else: +# +# ramdisk1 init, nsm.ko +# ramdisk2 cmd, env, rootfs/… +# +# `init` inserts the NSM driver, sends a heartbeat to the parent over vsock, +# then binds `/rootfs` onto itself, mounts the pseudo-filesystems *inside* it, +# and runs the contents of `/cmd` with the environment from `/env`. +# +# The mountpoints therefore live under `rootfs/`. An image without them dies +# with `mount: /dev: No such file or directory`, which reads like a bootstrap +# fault and is not. +{ lib +, stdenvNoCC +, cpio +, jq +, libfaketime +, closureInfo +, blobs +, eif-build +}: + +{ name + # Attribute set of destination path (relative to rootfs/) → store path, or a + # path in this repository. Each is interpolated, never `toString`ed: a path + # made a string that way carries no context, so the derivation does not + # depend on it — and a Nix that evaluates flakes lazily, as the one CI + # installs does, never puts the file in the store at all. +, payload + # Packages whose runtime closures must be inside the image. These are the + # derivations themselves, not the files copied out of them: `closureInfo` + # resolves store *paths*, and handing it `${pkg}/bin/thing` fails with + # "not in the Nix store", which is true and unhelpful. +, closureRoots ? [ ] + # What `init` executes, as an absolute path inside rootfs/. +, command + # Environment for the command, as an attribute set. +, env ? { } + # Copy the payload's runtime closure into the image. Needed for anything + # dynamically linked — which is everything except the static musl self-test. +, withClosure ? true + # Shell run inside the assembled rootfs/, for anything a plain file copy + # cannot express — symlinks, scripts, permissions. +, extraSetup ? "" +}: + +let + # The transitive runtime dependencies of the payload: glibc, libgcc and + # whatever else the binaries actually open at runtime. + closure = closureInfo { rootPaths = closureRoots; }; + + envFile = lib.concatStringsSep "\n" + (lib.mapAttrsToList (k: v: "${k}=${v}") env); + + copyPayload = lib.concatStringsSep "\n" (lib.mapAttrsToList + (dest: src: '' + mkdir -p "rootfs/$(dirname ${lib.escapeShellArg dest})" + cp -L ${lib.escapeShellArg "${src}"} "rootfs/${dest}" + chmod +x "rootfs/${dest}" || true + '') + payload); +in +stdenvNoCC.mkDerivation { + pname = "${name}-eif"; + version = "1"; + + dontUnpack = true; + nativeBuildInputs = [ cpio jq libfaketime eif-build ]; + + buildPhase = '' + runHook preBuild + + # ---- ramdisk 1: what boots ----------------------------------------- + mkdir -p rd1 + cp ${blobs}/blobs/x86_64/init rd1/init + cp ${blobs}/blobs/x86_64/nsm.ko rd1/nsm.ko + chmod +x rd1/init + + # ---- ramdisk 2: what runs ------------------------------------------ + mkdir -p rd2 && cd rd2 + mkdir -p rootfs/{dev/pts,dev/shm,proc,sys/fs/cgroup,run,tmp,etc} + + ${copyPayload} + + ${lib.optionalString withClosure '' + # Everything the payload links against, at the same store paths it was + # built to look for. `init` runs inside rootfs/, so the store has to be + # there rather than at the image root. + mkdir -p rootfs/nix/store + for path in $(cat ${closure}/store-paths); do + cp -a "$path" rootfs/nix/store/ + done + chmod -R u+w rootfs/nix/store + ''} + + ${lib.optionalString (extraSetup != "") '' + ( cd rootfs + ${extraSetup} + ) + ''} + + printf '%s\n' ${lib.escapeShellArg command} > cmd + printf '%s\n' ${lib.escapeShellArg envFile} > env + + cd .. + + # ---- the archives --------------------------------------------------- + # Reproducibility lives here, and it is not free. PCR0 and PCR1 are + # digests of exactly these bytes, so anything the archive records that + # varies between builds changes the measurement an enclave attests to. + # + # The `newc` header carries four such things, and all four had to be + # pinned. Without `--reproducible` the *bootstrap* ramdisk came out + # different every build even though its two input files never change, + # because cpio stores each file's inode and device numbers and those are + # whatever the filesystem handed out that time: + # + # inode, device --reproducible (renumbers inodes, drops devno) + # mtime touch -d @1; store paths are already epoch+1 + # uid, gid --owner=0:0 + # member order find | LC_ALL=C sort, which also keeps parent + # directories ahead of their contents + archive() { + local dir="$1" out="$2" + find "$dir" -mindepth 1 -exec touch -h -d @1 {} + + ( cd "$dir" && find . -mindepth 1 -printf '%P\0' | LC_ALL=C sort -z \ + | cpio --null -o -H newc --quiet --reproducible --owner=0:0 ) > "$out" + } + + archive rd1 ramdisk1.cpio + archive rd2 ramdisk2.cpio + + # ---- the image ------------------------------------------------------ + # `faketime` freezes the clock, because eif_build stamps wall-clock + # `BuildTime` into the image's metadata. Without it two builds of identical + # inputs differ by exactly 15 bytes: the 13-digit timestamp and the 4-byte + # header CRC it changes. + # + # Those bytes are in metadata, which no PCR covers, so the *measurements* + # were already reproducible without this. Freezing time makes the file + # itself byte-identical too, which matters for the plainer reason that + # people compare artifacts by hashing them, and because it lets + # `nix build --rebuild` be a check that passes rather than a known + # exception. + faketime -f '@1970-01-01 00:00:01' \ + eif_build \ + --kernel ${blobs}/blobs/x86_64/bzImage \ + --kernel_config ${blobs}/blobs/x86_64/bzImage.config \ + --cmdline "$(cat ${blobs}/blobs/x86_64/cmdline)" \ + --ramdisk ramdisk1.cpio \ + --ramdisk ramdisk2.cpio \ + --output ${name}.eif \ + | tee eif_build.log + + runHook postBuild + ''; + + installPhase = '' + runHook preInstall + + mkdir -p $out + cp ${name}.eif $out/${name}.eif + + # eif_build prints the measurements as JSON after a header line. Keeping + # them beside the image is what lets an operator compare a running + # enclave's attested PCR0 against the build that claims to have produced + # it — without that, a reproducible build proves nothing to anyone. + sed -n '/^{/,/^}/p' eif_build.log > $out/pcr.json + jq -e '.PCR0 | length == 96' $out/pcr.json > /dev/null \ + || { echo "eif_build did not report a PCR0; the image cannot be attested" >&2; exit 1; } + + echo "PCR0 $(jq -r .PCR0 $out/pcr.json)" + + runHook postInstall + ''; + + meta = { + description = "AWS Nitro Enclave image for ${name}"; + platforms = [ "x86_64-linux" ]; + }; +} diff --git a/runtime/Cargo.toml b/runtime/Cargo.toml new file mode 100644 index 0000000..b473dbf --- /dev/null +++ b/runtime/Cargo.toml @@ -0,0 +1,240 @@ +[package] +name = "enclave-runtime" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +description = "Runs a Wasm guest inside an AWS Nitro Enclave against a Merkle-anchored S3 filesystem" + +# Library plus binary. The library half was `s3fs-host` until it had exactly +# one consumer; the binary half is argument parsing over it. They are one crate +# because the split had stopped describing anything — `s3fs-host` had grown the +# NSM, the vsock tap device and the attestation endpoints, none of which mean +# anything outside an enclave. +# +# It stays a library rather than collapsing into `main.rs` because integration +# tests cannot import a binary-only crate, and `tests/serve_tls.rs` is where +# the certificate-binding property is actually checked. +[lib] +name = "enclave_runtime" +path = "src/lib.rs" + +[[bin]] +name = "enclave-runtime" +path = "src/main.rs" + +# The client half of the gate, for the QEMU harness. A shell script cannot sign +# a WebAuthn assertion, and the harness's self-signed certificate means no real +# platform authenticator would ever attest against it — so the e2e drives the +# gate with this. Behind `testing`, because it can mint assertions. +[[bin]] +name = "passkey-client" +path = "src/bin/passkey-client.rs" +required-features = ["testing"] + +# One feature, and it is not about size. `testing` compiles +# `auth::SoftwareAuthenticator`, which mints WebAuthn assertions — a forgery +# tool. It is off by default so it cannot reach a production image, and on for +# the QEMU harness, whose self-signed certificate means no real platform +# authenticator will ever attest against it. +# +# There used to be `aws` and `serve` features too, so a host embedding only the +# `wasi:filesystem` bindings could skip the AWS SDK and hyper. With one consumer +# that always wants both, they bought nothing and cost +# `--features aws,s3fs-host/serve` on every command. +[features] +default = [] +# `nitro-attestation/testing` costs nothing new: it is `dep:rcgen` plus +# `dep:time`, and rcgen is already enabled here by this very feature. It is +# what lets `testing::Enclave` mint documents a client can actually verify. +testing = ["dep:rcgen", "nitro-attestation/testing"] + +[dependencies] +s3fs-core = { path = "../crates/s3fs-core", features = ["aws"] } +nitro-nsm = { path = "../crates/nitro-nsm" } +nitro-attestation = { path = "../crates/nitro-attestation" } + +# The host compiles components, so it needs cranelift on top of the runtime. +wasmtime = { version = "=44.0.0", features = ["component-model", "async", "runtime", "cranelift", "std"] } +wasmtime-wasi = { workspace = true } +wasmtime-wasi-io = "=44.0.0" + +# `default-send-request` is deliberately OFF. With it on, `wasi:http`'s +# outgoing-handler quietly gains a working implementation — a guest could make +# arbitrary outbound requests, through a second rustls and its own root store, +# neither of which anything here configures. With it off, `send_request` +# becomes a trait method we are *required* to write, so refusing egress is a +# visible decision rather than a feature flag nobody looked at. See +# `serve::EgressPolicy`. +wasmtime-wasi-http = { version = "=44.0.0", default-features = false, features = ["p2"] } +hyper = { version = "1", features = ["http1", "http2", "server", "client"] } +# Protocol chosen per connection rather than per deployment. `auto::Builder` +# reads the HTTP/2 connection preface before deciding, so one listener serves +# both and a client that never heard of h2 is unaffected. h2 is what gRPC +# requires and nothing else here needs; it is pure Rust and brings no crypto of +# its own, so what it adds to the measured image is a framing layer. +hyper-util = { version = "0.1", default-features = false, features = ["server", "server-auto", "client", "client-legacy", "http1", "http2", "tokio"] } +http = "1" +http-body-util = "0.1" + +# aws-lc-rs, not ring: it is already the crypto stack for the block store's +# AEAD, signatures and key derivation, and a second one would mean two +# implementations of the same primitives inside one attested image. +# +# That argument has one accepted exception, below: `webauthn-rs` brings +# `openssl` and `openssl-sys`, which is a second stack *with C bindings*, inside +# the boundary PCR0 measures. It was taken knowingly, in exchange for not +# writing assertion verification by hand on the authentication path. Left +# unqualified this comment would contradict the manifest it is in. +rustls = { version = "0.23", default-features = false, features = ["std", "aws-lc-rs", "tls12"] } +# Already in the image via rustls, which parses every certificate it verifies. +# Naming it here adds no code to the build — only the ability to ask for the +# public key rather than re-deriving a parser we would have to trust. +webpki = { package = "rustls-webpki", version = "0.103" } +tokio-rustls = { version = "0.26", default-features = false, features = ["aws-lc-rs", "tls12"] } +# The client half, for `notify`: the runtime calls Firebase itself because the +# guest may not (see `serve::EgressPolicy`). Neither `aws-lc-rs` nor +# `webpki-roots` is enabled here on purpose — `notify::transport` builds its +# own `ClientConfig`, so the trust decision is in code a reviewer reads rather +# than behind a feature name nobody looked at. +hyper-rustls = { version = "0.27", default-features = false, features = ["http1", "http2"] } +# The roots that client validates against, compiled into the measured image +# rather than read from a file. Already in the tree via rustls-acme. +webpki-roots = "1" +# Self-signed certificates, for the test harnesses and the QEMU emulator only. +# Optional, and off in a production build: the runtime serves ACME-issued +# certificates, so a certificate generator has no business inside the boundary +# PCR0 measures. +rcgen = { version = "0.14", default-features = false, features = ["aws_lc_rs", "pem"], optional = true } +# TLS-ALPN-01, so the challenge arrives on the port the parent already +# forwards. `webpki-roots` is not optional in practice: reaching the ACME +# directory means validating the CA's own TLS certificate, and the enclave +# cannot borrow the parent's trust store — the parent is the party being +# excluded. +# PEM, not a hand-rolled split: the certificate half of rustls-acme's blob is +# the ACME directory's HTTP response verbatim, so its line endings are the CA's +# choice and CRLF is as valid as LF. Already in the tree via rcgen/rustls-acme. +pem = "3" +rustls-acme = { version = "0.15", default-features = false, features = ["tokio", "aws-lc-rs", "tls12", "webpki-roots"] } +# Sealing the ACME cache uses the same AES-256-GCM as the block store. +aws-lc-rs = { workspace = true } +blake3 = { workspace = true } +# rustls-acme drives issuance through a Stream. +futures = { workspace = true } +# The state_root pre-image and the receipt payloads. CBOR because the +# receipt travels in an attestation document, which is CBOR already. +ciborium = "0.2" +base64 = "0.22" +# Scrubs the FCM service-account key on drop. Already in the tree via +# aws-credential-types; naming it here is what lets `notify::oauth` hold +# the key in a `Zeroizing` rather than a bare String. +zeroize = { workspace = true } +hex = "0.4" + +# The PTP clock is read through the POSIX dynamic-clock mechanism. rustix is +# already compiled into the tree via wasmtime, so this adds no build cost and +# avoids a direct libc dependency. +rustix = { version = "1", features = ["time", "fs"] } +# Pinned to whatever `wasmtime-wasi` accepts for `secure_random`, which is +# cap-rand's re-export of rand 0.8 — not a version to guess at. +cap-rand = "3" +getrandom = "0.3" + +clap = { version = "4", features = ["derive", "env"] } +anyhow = { workspace = true } +async-trait = { workspace = true } +bytes = { workspace = true } +tokio = { workspace = true, features = ["sync", "rt", "rt-multi-thread", "macros", "signal", "net", "io-util", "time"] } +tracing = { workspace = true } +tracing-subscriber = { version = "0.3", features = ["env-filter", "fmt"] } +# CiphertextForRecipient — what KMS returns when the response is encrypted to +# the enclave rather than sent as plaintext — is a CMS EnvelopedData (RFC 5652). +# The crate parses the structure; `keys::recipient` decides what shape is +# acceptable, which is deliberately one shape and no other. +cms = { version = "0.2.3", default-features = false, features = ["alloc"] } +# Already pulled in by `cms`; named here for the OIDs and the OCTET STRING the +# unwrap has to read out of the algorithm parameters. +der = { version = "0.7", default-features = false, features = ["alloc", "oid", "std"] } +# The master secret is minted and released by KMS, never by configuration. +# `Recipient` is the parameter that matters: it makes KMS encrypt its answer to +# a key inside the enclave and omit `Plaintext` entirely, so the parent can +# proxy the call without being able to read it. SSM holds the resulting +# ciphertext, and nothing else. +aws-sdk-kms = { version = "1.51.0", default-features = false, features = ["rt-tokio", "rustls", "behavior-version-latest"] } +aws-sdk-ssm = { version = "1.56.0", default-features = false, features = ["rt-tokio", "rustls", "behavior-version-latest"] } +# Guest stdout and stderr, out to CloudWatch Logs from inside the enclave. +# +# Yes, this puts an AWS SDK inside the boundary PCR0 measures — but KMS, SSM and +# S3 are already here, for the same reason: an enclave with no network of its +# own still calls AWS over gvproxy. It brings no new crypto stack; kms and ssm +# already pull `rustls`, which is what the ban above is actually about. +aws-sdk-cloudwatchlogs = { version = "1.131.0", default-features = false, features = ["rt-tokio", "rustls", "behavior-version-latest"] } +# The default credential chain, which resolves through gvproxy to the parent +# instance's IMDS. Named explicitly rather than relied on as a fallback, so it +# is obvious *which* identity guest logs are written with. +aws-config = { workspace = true, features = ["behavior-version-latest", "rt-tokio", "rustls"] } +# The feature adds no code beyond one registration constructor — see +# `auth::routes`. Without it, options ask for a non-discoverable credential, +# which Android below 14 creates outside Password Manager where nothing finds it. +webauthn-rs = { version = "0.5.5", features = ["workaround-google-passkey-specific-issues"] } +serde_json = "1.0.151" +serde = { version = "1.0.229", features = ["derive"] } + +[dev-dependencies] +# `StaticReplayClient`: canned HTTP responses for the CloudWatch client, so the +# tests assert the request that is actually serialised rather than a mock of +# one. A dev-dependency, so none of it reaches the image. It is already in the +# tree as a transitive dependency of the SDK — naming it here adds no code to +# the build, only the ability to ask for its test utilities. +aws-smithy-http-client = { version = "1", features = ["test-util"] } +# `start_paused`: a timeout test should not spend the timeout. +tokio = { workspace = true, features = ["macros", "rt", "rt-multi-thread", "test-util"] } +aws-smithy-types = { workspace = true } +# Building canned HTTP requests/responses for the replay client. +http = "1" +# The HTTP/2 client half, for the streaming suites. Driving h2 from a real +# client is the only way to prove the server negotiates it and frames it +# correctly. +# +# This used to say the runtime never calls out over HTTP and that `client` was +# dev-only. Neither was true: `aws-config`'s `rustls` feature has pulled a hyper +# client into the production graph for as long as the AWS SDKs have been here, +# and `notify` now uses one deliberately. +hyper = { version = "1", features = ["client", "http1", "http2", "server"] } +hyper-util = { version = "0.1", default-features = false, features = ["client", "client-legacy", "server", "server-auto", "http1", "http2", "tokio"] } +# Turning a channel into a request body, so a test can hold the request open +# and feed it one gRPC frame at a time — which is the only way to observe the +# guest answering before the client has finished asking. +tokio-stream = "0.1" +# A real gRPC client, so the guest's framing is checked against an +# implementation that did not come from this repository. Hand-rolled framing on +# both ends would only prove the two halves agree with each other. +# +# Dev-only, and it has to be: `tonic` does not build for wasm32-wasip2, which +# is why the *guest* frames gRPC by hand in the first place. +tonic = { version = "0.14", default-features = false, features = ["codegen"] } +tonic-prost = "0.14" +prost = "0.14" +tower = { version = "0.5", features = ["util"] } +# The integration suites drive the gate with a software passkey, which only +# exists under `testing`. A dev-dependency on the crate itself is how a crate +# turns a feature on for its own tests without turning it on for consumers. +enclave-runtime = { path = ".", features = ["testing"] } +# Checking that a generated certificate really parses as X.509. +x509-parser = "0.18" +nitro-nsm = { path = "../crates/nitro-nsm", features = ["testing"] } +# Builds genuinely signed attestation documents for the TLS binding tests. +nitro-attestation = { path = "../crates/nitro-attestation", features = ["testing"] } +# The fresh-versus-warm instance question: how much of a request is +# instantiation, and is it enough to justify pooling instances per client. +criterion = { version = "0.5", features = ["async_tokio"] } +# Building the CMS envelope a test has to unwrap. Both arrive with `cms` +# anyway; naming them here keeps the test encoder honest rather than reaching +# through a re-export. +spki = "0.7" +x509-cert = "0.2" + +[[bench]] +name = "instance_cost" +harness = false diff --git a/runtime/benches/instance_cost.rs b/runtime/benches/instance_cost.rs new file mode 100644 index 0000000..cbe885a --- /dev/null +++ b/runtime/benches/instance_cost.rs @@ -0,0 +1,201 @@ +//! What a fresh guest instance actually costs, and what a request costs around +//! it. +//! +//! This benchmark exists to answer one question with a number rather than an +//! intuition: **is per-client instance pooling worth its complexity?** Pooling +//! saves exactly one thing — the instantiation in the middle of every request — +//! and buys it at the price of per-client lifecycle, eviction, and a +//! concurrency model where two clients run at once. That trade is only worth +//! making if instantiation is a large fraction of a request. +//! +//! So the interesting output is not any single line, it is the ratio: +//! +//! ```text +//! instantiate ÷ dispatch/committing → what pooling could remove +//! ``` +//! +//! Against [`MemoryBackend`] there is no network, so `dispatch/committing` +//! measures the engine's own cost — encryption, hashing, the copy-on-write +//! rebuild, the commit. Against real S3 the round trips are added to the +//! denominator and never to the numerator, so **the ratio measured here is the +//! most favourable case pooling will ever see.** If it is small here, it is +//! smaller in production. +//! +//! ```bash +//! (cd examples/guest-http && cargo build --release --target wasm32-wasip2) +//! cargo bench -p enclave-runtime +//! ``` + +use std::path::PathBuf; +use std::sync::Arc; + +use criterion::{criterion_group, criterion_main, Criterion}; +use tokio::runtime::Runtime; + +use bytes::Bytes; +use enclave_runtime::{GuestEnvironment, HostClock, ServeHandle}; +use http_body_util::{BodyExt, Full}; +use s3fs_core::backend::memory::MemoryBackend; +use s3fs_core::backend::Backend; +use s3fs_core::{Config, Fs, MasterSecret}; +use wasmtime_wasi_http::p2::bindings::http::types::Scheme; +use wasmtime_wasi_http::p2::body::HyperOutgoingBody; + +fn runtime() -> Runtime { + tokio::runtime::Builder::new_multi_thread() + .worker_threads(2) + .enable_all() + .build() + .expect("tokio runtime") +} + +fn component_path() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-http/target/wasm32-wasip2/release/guest-http.wasm") +} + +fn component_bytes() -> Vec { + std::fs::read(component_path()).unwrap_or_else(|e| { + panic!( + "reading {}: {e}\nbuild it first: (cd examples/guest-http && \ + cargo build --release --target wasm32-wasip2)", + component_path().display() + ) + }) +} + +/// A handle over a memory-backed filesystem, compiled once. +/// +/// Compilation and `instantiate_pre` are startup costs the runtime pays once +/// for the life of the process, so they belong in setup — measuring them here +/// would drown the per-request number this benchmark exists to find. +async fn handle() -> ServeHandle { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([7u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("creating the memory-backed filesystem"); + + // Detached deliberately: the collector runs for as long as this + // environment can send, which is what a test wants. Production drains it + // explicitly instead. + let (logs, _collector) = + enclave_runtime::guest_io::start(std::sync::Arc::new(enclave_runtime::TracingLogSink)); + let guest = GuestEnvironment::new( + fs, + Box::new(HostClock), + Arc::new(nitro_nsm::fake::FakeNsm::new()), + &[], + &[], + logs, + ) + .expect("guest environment"); + + let engine = ServeHandle::engine_with_watchdog().expect("engine"); + ServeHandle::new(&engine, &component_bytes(), guest).expect("preparing the guest") +} + +fn body() -> HyperOutgoingBody { + Full::new(Bytes::new()) + .map_err(|e: std::convert::Infallible| match e {}) + .boxed_unsync() +} + +async fn get(handle: &ServeHandle, path: &str) { + let req = hyper::Request::builder() + .method("GET") + .uri(format!("http://enclave.test{path}")) + .body(body()) + .expect("well-formed request"); + let resp = handle + .handle(Scheme::Http, req, None) + .await + .expect("guest handled the request"); + // Drained, not dropped: the head returns before the guest has finished, so + // stopping at the head would time half a request. + resp.into_body().collect().await.expect("collecting body"); +} + +/// A filesystem that already exists, so it can be mounted rather than created. +async fn existing() -> (Arc, Arc) { + let data = Arc::new(MemoryBackend::new()) as Arc; + let roots = Arc::new(MemoryBackend::new()) as Arc; + Fs::create( + data.clone(), + roots.clone(), + &MasterSecret::from_bytes([7u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("creating the filesystem"); + (data, roots) +} + +fn bench(c: &mut Criterion) { + let rt = runtime(); + let handle = rt.block_on(handle()); + + let mut group = c.benchmark_group("guest"); + + // The numerator. Exactly what a warm pool would skip: a fresh `Store` and + // an `instantiate_async`, with no request dispatched through it. + group.bench_function("instantiate", |b| { + b.to_async(&rt) + .iter(|| async { handle.verify_instantiates().await.expect("instantiates") }); + }); + + // A whole request that touches nothing. Instantiation plus dispatch plus + // the guest's own trivial work — the floor for any request at all. + group.bench_function("dispatch/trivial", |b| { + b.to_async(&rt).iter(|| get(&handle, "/memory")); + }); + + // A whole request that reads, modifies, writes and commits — one + // transaction group, one root record. This is the denominator that matters, + // because it is the shape of real work, and against real S3 it grows while + // `instantiate` does not. + group.bench_function("dispatch/committing", |b| { + b.to_async(&rt).iter(|| get(&handle, "/counter")); + }); + + // What a *per-client* filesystem would cost to bring up: derive its key + // material, read its signed root record, verify the signature, open the + // object set. Against `MemoryBackend` this is the CPU floor and nothing + // else — in production the root record and the object set are reads against + // S3, so the real figure is this plus round trips. + // + // Worth measuring beside `instantiate`, because if instances are per client + // then so is this, and it is the larger of the two by orders of magnitude. + // Whatever a per-client design keeps warm, this is the thing worth keeping. + let rt2 = runtime(); + let (data, roots) = rt2.block_on(existing()); + group.bench_function("mount", |b| { + b.to_async(&rt2).iter(|| { + let data = data.clone(); + let roots = roots.clone(); + async move { + Fs::mount( + data, + roots, + &MasterSecret::from_bytes([7u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + None, + ) + .await + .expect("mounting") + } + }); + }); + + group.finish(); +} + +criterion_group!(benches, bench); +criterion_main!(benches); diff --git a/runtime/src/auth/authenticator.rs b/runtime/src/auth/authenticator.rs new file mode 100644 index 0000000..887dd18 --- /dev/null +++ b/runtime/src/auth/authenticator.rs @@ -0,0 +1,248 @@ +//! A passkey in software, for tests. +//! +//! Every property the gate has — origin, RP ID, user verification, the +//! signature, the binding to one request — can only be tested against +//! something that produces real assertions. A phone cannot be driven from a +//! test suite, so this is a P-256 key that emits exactly the bytes a phone +//! would: the same `clientDataJSON`, the same authenticator-data layout, the +//! same ECDSA signature over the same message. +//! +//! It is a fixture, not a fake. Nothing here is more permissive than a real +//! authenticator — the runtime's verification does not know the difference and +//! is not told. What it *can* do that a phone cannot is misbehave on purpose: +//! sign the wrong challenge, clear the user-verification flag, claim another +//! origin. Those are the tests worth having, and they need an authenticator +//! that will do as it is told. +//! +//! Compiled into the crate rather than a test file so the QEMU harness can use +//! it too — its self-signed certificate means no real platform authenticator +//! will ever attest against it, so this is the only way that leg exercises the +//! gate at all. + +use aws_lc_rs::rand::SystemRandom; +use aws_lc_rs::signature::{EcdsaKeyPair, KeyPair, ECDSA_P256_SHA256_ASN1_SIGNING}; +use base64::Engine as _; + +/// Authenticator-data flags, as the spec names them. +pub mod flags { + /// User present — someone touched it. + pub const UP: u8 = 0x01; + /// User verified — biometric or PIN, not merely presence. + pub const UV: u8 = 0x04; + /// Attested credential data follows. Registration only. + pub const AT: u8 = 0x40; +} + +fn b64(bytes: &[u8]) -> String { + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(bytes) +} + +/// A software passkey bound to one relying party. +pub struct SoftwareAuthenticator { + key: EcdsaKeyPair, + /// The private key in PKCS#8, kept so a caller can persist and restore the + /// passkey. A real authenticator's key never leaves its hardware; this one + /// has to, because a shell script invokes the client afresh for every + /// request and a passkey that vanished between them could not be used at + /// all. It is a test fixture — it is behind a feature for exactly this + /// kind of reason. + key_pkcs8: Vec, + credential_id: Vec, + rp_id: String, + /// Atomic rather than a `Cell`: a real authenticator can be asked for two + /// assertions at once, and a fixture that could not be shared across tasks + /// would rule out testing exactly that. + counter: std::sync::atomic::AtomicU32, +} + +impl SoftwareAuthenticator { + pub fn new(rp_id: &str) -> Self { + let key_pkcs8 = + EcdsaKeyPair::generate_pkcs8(&ECDSA_P256_SHA256_ASN1_SIGNING, &SystemRandom::new()) + .expect("generating a P-256 key"); + let mut credential_id = vec![0u8; 32]; + aws_lc_rs::rand::fill(&mut credential_id).expect("credential id"); + Self::restore(rp_id, &credential_id, key_pkcs8.as_ref(), 0) + .expect("a key we just generated parses") + } + + /// Rebuild a passkey from a credential id, its private key, and the + /// signature counter it had reached. + /// + /// The counter has to come back too. An authenticator that restarted at + /// zero would present a count no higher than the one registration + /// recorded, and a relying party is entitled to read that as a cloned + /// credential — which is exactly what it is protecting against. + pub fn restore( + rp_id: &str, + credential_id: &[u8], + key_pkcs8: &[u8], + counter: u32, + ) -> anyhow::Result { + let key = EcdsaKeyPair::from_pkcs8(&ECDSA_P256_SHA256_ASN1_SIGNING, key_pkcs8) + .map_err(|e| anyhow::anyhow!("restoring a passkey: {e}"))?; + Ok(SoftwareAuthenticator { + key, + key_pkcs8: key_pkcs8.to_vec(), + credential_id: credential_id.to_vec(), + rp_id: rp_id.to_string(), + counter: std::sync::atomic::AtomicU32::new(counter), + }) + } + + /// The counter this passkey has reached, for a caller that persists it. + pub fn counter(&self) -> u32 { + self.counter.load(std::sync::atomic::Ordering::Relaxed) + } + + pub fn credential_id(&self) -> &[u8] { + &self.credential_id + } + + /// The private key, for a caller that has to persist this passkey. + pub fn private_key_pkcs8(&self) -> &[u8] { + &self.key_pkcs8 + } + + /// The credential public key, as the COSE_Key an authenticator reports. + /// + /// `{1: 2 (EC2), 3: -7 (ES256), -1: 1 (P-256), -2: x, -3: y}`, which is + /// what `attestedCredentialData` carries and what the relying party stores. + fn cose_key(&self) -> Vec { + let point = self.key.public_key().as_ref(); + assert_eq!(point.len(), 65, "expected an uncompressed P-256 point"); + assert_eq!(point[0], 0x04, "expected an uncompressed P-256 point"); + let value = ciborium::Value::Map(vec![ + ( + ciborium::Value::Integer(1.into()), + ciborium::Value::Integer(2.into()), + ), + ( + ciborium::Value::Integer(3.into()), + ciborium::Value::Integer((-7).into()), + ), + ( + ciborium::Value::Integer((-1).into()), + ciborium::Value::Integer(1.into()), + ), + ( + ciborium::Value::Integer((-2).into()), + ciborium::Value::Bytes(point[1..33].to_vec()), + ), + ( + ciborium::Value::Integer((-3).into()), + ciborium::Value::Bytes(point[33..65].to_vec()), + ), + ]); + let mut out = Vec::new(); + ciborium::into_writer(&value, &mut out).expect("encoding a COSE key"); + out + } + + /// `rpIdHash ‖ flags ‖ counter`, plus attested credential data when + /// registering. + fn authenticator_data(&self, flags: u8, attested: bool) -> Vec { + let mut data = nitro_attestation::sha256(self.rp_id.as_bytes()).to_vec(); + data.push(flags); + let counter = self + .counter + .fetch_add(1, std::sync::atomic::Ordering::Relaxed) + + 1; + data.extend_from_slice(&counter.to_be_bytes()); + if attested { + data.extend_from_slice(&[0u8; 16]); // AAGUID: zero, as platform authenticators report + data.extend_from_slice(&(self.credential_id.len() as u16).to_be_bytes()); + data.extend_from_slice(&self.credential_id); + data.extend_from_slice(&self.cose_key()); + } + data + } + + fn client_data(&self, kind: &str, challenge: &str, origin: &str) -> Vec { + // Field order is the authenticator's to choose and the relying party + // must not depend on it, so this is not the same order webauthn-rs + // writes. That is deliberate: it is one more thing the verifier is + // being checked for not assuming. + format!( + r#"{{"type":"{kind}","challenge":"{challenge}","origin":"{origin}","crossOrigin":false}}"# + ) + .into_bytes() + } + + fn sign(&self, authenticator_data: &[u8], client_data: &[u8]) -> Vec { + let mut message = authenticator_data.to_vec(); + message.extend_from_slice(&nitro_attestation::sha256(client_data)); + self.key + .sign(&SystemRandom::new(), &message) + .expect("signing") + .as_ref() + .to_vec() + } + + /// A registration response, attestation format `none`. + /// + /// `none` is what a platform authenticator produces for a passkey unless + /// the relying party asks otherwise, so it is what the runtime has to + /// accept. + pub fn register(&self, challenge: &str, origin: &str) -> serde_json::Value { + let client_data = self.client_data("webauthn.create", challenge, origin); + let auth_data = self.authenticator_data(flags::UP | flags::UV | flags::AT, true); + + let attestation = ciborium::Value::Map(vec![ + ( + ciborium::Value::Text("fmt".into()), + ciborium::Value::Text("none".into()), + ), + ( + ciborium::Value::Text("attStmt".into()), + ciborium::Value::Map(vec![]), + ), + ( + ciborium::Value::Text("authData".into()), + ciborium::Value::Bytes(auth_data), + ), + ]); + let mut attestation_object = Vec::new(); + ciborium::into_writer(&attestation, &mut attestation_object).expect("attestation object"); + + serde_json::json!({ + "id": b64(&self.credential_id), + "rawId": b64(&self.credential_id), + "type": "public-key", + "extensions": {}, + "response": { + "attestationObject": b64(&attestation_object), + "clientDataJSON": b64(&client_data), + } + }) + } + + /// An assertion, exactly as a phone would produce one. + pub fn assert(&self, challenge: &str, origin: &str) -> serde_json::Value { + self.assert_with(challenge, origin, flags::UP | flags::UV) + } + + /// An assertion with the flags chosen by the caller. + /// + /// Clearing [`flags::UV`] is how a test asks "what if the user merely had + /// an unlocked phone in their pocket" — which must be refused, because a + /// cosigner's approval is supposed to mean a person did something. + pub fn assert_with(&self, challenge: &str, origin: &str, flags: u8) -> serde_json::Value { + let client_data = self.client_data("webauthn.get", challenge, origin); + let auth_data = self.authenticator_data(flags, false); + let signature = self.sign(&auth_data, &client_data); + + serde_json::json!({ + "id": b64(&self.credential_id), + "rawId": b64(&self.credential_id), + "type": "public-key", + "extensions": {}, + "response": { + "authenticatorData": b64(&auth_data), + "clientDataJSON": b64(&client_data), + "signature": b64(&signature), + "userHandle": null, + } + }) + } +} diff --git a/runtime/src/auth/challenge.rs b/runtime/src/auth/challenge.rs new file mode 100644 index 0000000..bca7e88 --- /dev/null +++ b/runtime/src/auth/challenge.rs @@ -0,0 +1,258 @@ +//! Challenges, and what each one is *for*. +//! +//! A WebAuthn assertion proves that a passkey signed a challenge with user +//! verification. It says nothing at all about which HTTP request that was +//! meant to authorize — the protocol has no field for it, and an authenticator +//! displays nothing the user could check. +//! +//! So the binding is the server's job, and this is where it is kept. Issuing a +//! challenge records the exact request it was issued for; consuming one hands +//! that record back, and the caller compares it against the request that +//! actually arrived. An assertion moved to a different body, a different path +//! or a different method names a challenge whose record does not match, and is +//! refused. +//! +//! Three properties make that hold, and all three are enforced here rather than +//! left to a caller to remember: +//! +//! - **Single use.** [`ChallengeStore::consume`] removes the entry as it +//! returns it, under one lock, so two copies of one assertion cannot both +//! find it. That is not a nicety: replaying a signing approval is the whole +//! attack. +//! - **Short lived.** An unconsumed challenge expires, so an assertion captured +//! and held is worthless within the minute. +//! - **Bounded.** The table is capped and swept, because anyone who can reach +//! the runtime can ask for challenges, and memory in an enclave is finite. +//! +//! Nothing here survives a restart, and nothing should: the browser asks for a +//! new challenge, and every assertion in flight becomes worthless. That is the +//! safe direction to fail in. + +use std::collections::HashMap; +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +use webauthn_rs::prelude::PasskeyAuthentication; + +use super::token::InteractionScope; + +/// How long a challenge is good for. +/// +/// Long enough for a human to look at a prompt and present a finger; short +/// enough that a captured assertion is stale before it can be used. WebAuthn +/// gives the authenticator its own timeout, but that one is advisory and the +/// client controls it — this one is not. +pub const DEFAULT_TTL: Duration = Duration::from_secs(60); + +/// Outstanding challenges allowed at once. +/// +/// Issuing one is unauthenticated by necessity — a client cannot prove who it +/// is until it has a challenge to sign — so this is the bound on what an +/// anonymous caller can make the runtime hold. +pub const DEFAULT_CAPACITY: usize = 1024; + +/// A challenge that has been issued and not yet used. +struct Pending { + scope: InteractionScope, + /// The crate's own state for this challenge. Holding it here is what makes + /// the challenge single-use: it exists in exactly one place, and consuming + /// takes it. + state: PasskeyAuthentication, + expires: Instant, +} + +/// Why a challenge could not be used. +/// +/// Distinguished for the log, never for the client: an attacker learns +/// something from "expired" versus "already used" and a legitimate caller +/// learns nothing it can act on. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ChallengeError { + Unknown, + Expired, + Full, +} + +impl std::fmt::Display for ChallengeError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + ChallengeError::Unknown => "no such challenge, or it has already been used", + ChallengeError::Expired => "the challenge expired", + ChallengeError::Full => "too many outstanding challenges", + }) + } +} + +pub struct ChallengeStore { + pending: Mutex>, + ttl: Duration, + capacity: usize, +} + +impl ChallengeStore { + pub fn new(ttl: Duration, capacity: usize) -> Self { + ChallengeStore { + pending: Mutex::new(HashMap::new()), + ttl, + capacity: capacity.max(1), + } + } + + /// Record a challenge and what it is for. + /// + /// `id` is generated by the caller from the enclave's entropy source, so + /// the identifier a client quotes back is unguessable — otherwise an + /// attacker could name somebody else's outstanding challenge and race them + /// for it. + pub fn issue( + &self, + id: [u8; 16], + scope: InteractionScope, + state: PasskeyAuthentication, + ) -> Result<(), ChallengeError> { + let now = Instant::now(); + let mut pending = self.pending.lock().expect("challenge store poisoned"); + pending.retain(|_, p| p.expires > now); + if pending.len() >= self.capacity { + return Err(ChallengeError::Full); + } + pending.insert( + id, + Pending { + scope, + state, + expires: now + self.ttl, + }, + ); + Ok(()) + } + + /// Take a challenge, if it is there and still good. + /// + /// Removes it whether or not it turns out to verify. A challenge that was + /// offered to a bad assertion has been seen by whoever sent it, and giving + /// them a second attempt at the same one buys nothing but attempts. + pub fn consume( + &self, + id: &[u8; 16], + ) -> Result<(InteractionScope, PasskeyAuthentication), ChallengeError> { + let mut pending = self.pending.lock().expect("challenge store poisoned"); + let entry = pending.remove(id).ok_or(ChallengeError::Unknown)?; + if entry.expires <= Instant::now() { + return Err(ChallengeError::Expired); + } + Ok((entry.scope, entry.state)) + } + + pub fn outstanding(&self) -> usize { + let now = Instant::now(); + let mut pending = self.pending.lock().expect("challenge store poisoned"); + pending.retain(|_, p| p.expires > now); + pending.len() + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::auth::testing::Relying; + + /// Issued, then used once — and the second attempt finds nothing. This is + /// what stops a captured signing approval being replayed. + #[test] + fn a_challenge_can_be_consumed_exactly_once() { + let rp = Relying::new(); + let store = ChallengeStore::new(DEFAULT_TTL, DEFAULT_CAPACITY); + let id = [1u8; 16]; + let binding = InteractionScope::new("POST", "/sign", None); + + store + .issue(id, binding.clone(), rp.begin_authentication().1) + .unwrap(); + assert_eq!(store.consume(&id).unwrap().0, binding); + assert_eq!(store.consume(&id).unwrap_err(), ChallengeError::Unknown); + } + + #[test] + fn an_expired_challenge_is_refused() { + let rp = Relying::new(); + let store = ChallengeStore::new(Duration::from_millis(1), DEFAULT_CAPACITY); + let id = [2u8; 16]; + store + .issue( + id, + InteractionScope::new("POST", "/sign", None), + rp.begin_authentication().1, + ) + .unwrap(); + std::thread::sleep(Duration::from_millis(20)); + assert_eq!(store.consume(&id).unwrap_err(), ChallengeError::Expired); + } + + #[test] + fn an_unknown_challenge_is_refused() { + let store = ChallengeStore::new(DEFAULT_TTL, DEFAULT_CAPACITY); + assert_eq!( + store.consume(&[9u8; 16]).unwrap_err(), + ChallengeError::Unknown + ); + } + + /// Issuing is unauthenticated by necessity, so the table has to be bounded + /// against whoever can reach the port. + #[test] + fn the_store_refuses_to_grow_without_bound() { + let rp = Relying::new(); + let store = ChallengeStore::new(DEFAULT_TTL, 4); + for i in 0..4u8 { + let mut id = [0u8; 16]; + id[0] = i; + store + .issue( + id, + InteractionScope::new("GET", "/", None), + rp.begin_authentication().1, + ) + .unwrap(); + } + assert_eq!( + store + .issue( + [99u8; 16], + InteractionScope::new("GET", "/", None), + rp.begin_authentication().1 + ) + .unwrap_err(), + ChallengeError::Full + ); + assert_eq!(store.outstanding(), 4); + } + + /// Expired entries are swept, so a burst does not wedge the store for the + /// lifetime of the process. + #[test] + fn expiry_makes_room() { + let rp = Relying::new(); + let store = ChallengeStore::new(Duration::from_millis(1), 2); + for i in 0..2u8 { + let mut id = [0u8; 16]; + id[0] = i; + store + .issue( + id, + InteractionScope::new("GET", "/", None), + rp.begin_authentication().1, + ) + .unwrap(); + } + std::thread::sleep(Duration::from_millis(20)); + assert_eq!(store.outstanding(), 0); + store + .issue( + [7u8; 16], + InteractionScope::new("GET", "/", None), + rp.begin_authentication().1, + ) + .unwrap(); + } +} diff --git a/runtime/src/auth/credential.rs b/runtime/src/auth/credential.rs new file mode 100644 index 0000000..ab09d04 --- /dev/null +++ b/runtime/src/auth/credential.rs @@ -0,0 +1,534 @@ +//! Registered passkeys, and the tenants they speak for. +//! +//! Records live in `/runtime/credentials/` in the shared filesystem — **above +//! every tenant's scope**, so no guest can name them. That is not a convention +//! here: `974dac2` made the resolver refuse every path that leaves a tenant's +//! subtree, so a guest asking for `/runtime/credentials/…` gets its own +//! `runtime/credentials/…`, which does not exist. +//! +//! Being in the filesystem at all is what makes them trustworthy. The tree is +//! covered by one signed root record and that record is attested by the boot +//! receipt, so a host cannot add a credential, repoint one at another tenant, +//! or un-revoke one without breaking a signature the enclave checks before it +//! serves anything. +//! +//! ## The tenant id is minted, not derived +//! +//! Thirty-two random bytes from the NSM at first registration, stored beside +//! the credential. Deriving it from the credential would have been simpler and +//! would have been wrong: a tenant is a *person*, and a person has a phone, a +//! tablet and a hardware key. Derivation would tie the identity to one of +//! them, so adding a backup passkey would silently create a second tenant with +//! an empty filesystem — the worst possible outcome dressed as a feature. +//! +//! ## Why the sign counter is not persisted per request +//! +//! The counter exists to spot a cloned authenticator. Writing it on every +//! request would put a filesystem transaction — and in production a commit to +//! S3 — in the path of every signature, to record a number that platform +//! passkeys synced through iCloud or Google report as zero forever. +//! +//! So it is held in memory, checked for regression while the process lives, +//! and logged when it moves backwards. A restart forgets it, which loses the +//! ability to detect a clone *across* restarts and costs nothing else. That is +//! a deliberate trade rather than an omission; if counters ever become +//! meaningful for the authenticators in use, persisting them is a small change +//! and this comment is where to start. + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +use anyhow::{Context, Result}; +use s3fs_core::{Fs, FsError, Inode, OpenFlags}; +use serde::{Deserialize, Serialize}; +use webauthn_rs::prelude::Passkey; + +use crate::auth::gate::{CredentialRecord, CredentialStore}; +use crate::tenant::ensure_dir; + +/// Runtime-owned state, above every tenant scope. +pub const RUNTIME_DIR: &str = "runtime"; +/// Registered credentials, one file each. +pub const CREDENTIALS_DIR: &str = "credentials"; + +/// Ceiling on a record. A `Passkey` is a few hundred bytes; this is loose +/// enough never to bite and tight enough that a corrupt entry cannot be read +/// into unbounded memory. +const MAX_RECORD_BYTES: usize = 8 * 1024; + +/// A credential as it is stored. +/// +/// Versioned because it is on-disk and outlives the code that wrote it: an +/// unknown version has to be a refusal rather than a misparse of key material. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct StoredCredential { + pub version: u32, + pub tenant_id: [u8; 16], + pub passkey: Passkey, + /// Revoked credentials are kept, not deleted. A lost phone's key must be + /// refused *by name*; forgetting it would mean the same credential could + /// simply register again. + pub active: bool, + pub created_ms: u64, + pub counter: u32, +} + +const RECORD_VERSION: u32 = 1; + +/// Credentials in the shared filesystem, with the volatile parts in memory. +pub struct FilesystemCredentials { + fs: Arc, + /// Records read from the filesystem, plus the counters that are not + /// written back. Also the invalidation point: revocation updates both. + cache: Mutex, CredentialRecord>>, +} + +impl FilesystemCredentials { + pub fn new(fs: Arc) -> Self { + FilesystemCredentials { + fs, + cache: Mutex::new(HashMap::new()), + } + } + + async fn dir(&self) -> Result> { + let root = self.fs.root(); + let runtime = ensure_dir(&self.fs, &root, RUNTIME_DIR).await?; + ensure_dir(&self.fs, &runtime, CREDENTIALS_DIR).await + } + + fn name(credential_id: &[u8]) -> String { + hex::encode(credential_id) + } + + /// Write a credential, refusing to overwrite one that exists. + /// + /// `create_new`, so a second registration of the same credential id is an + /// error rather than a silent repointing — which is what taking over + /// somebody else's tenant would look like. + pub async fn register(&self, credential_id: &[u8], record: StoredCredential) -> Result<()> { + let dir = self.dir().await?; + let path = format!( + "/{RUNTIME_DIR}/{CREDENTIALS_DIR}/{}", + Self::name(credential_id) + ); + let _ = dir; + + let mut bytes = Vec::new(); + ciborium::into_writer(&record, &mut bytes).context("encoding a credential record")?; + anyhow::ensure!( + bytes.len() <= MAX_RECORD_BYTES, + "credential record is {} bytes, over the {MAX_RECORD_BYTES} limit", + bytes.len() + ); + + let handle = self + .fs + .open(&path, OpenFlags::create_new()) + .await + .map_err(|e| anyhow::anyhow!(e)) + .with_context(|| format!("creating {path}"))?; + self.fs + .pwrite(&handle, 0, &bytes) + .await + .map_err(|e| anyhow::anyhow!(e)) + .context("writing a credential record")?; + // Committed before the caller is told it is registered: a registration + // that had not reached the store would be a passkey the user believes + // works and the next boot has never heard of. + self.fs + .sync(&handle) + .await + .map_err(|e| anyhow::anyhow!(e)) + .context("committing a credential record")?; + self.fs + .close(&handle) + .await + .map_err(|e| anyhow::anyhow!(e)) + .context("closing a credential record")?; + + self.cache.lock().expect("credentials poisoned").insert( + credential_id.to_vec(), + CredentialRecord { + tenant_id: record.tenant_id, + passkey: record.passkey, + active: record.active, + counter: record.counter, + }, + ); + Ok(()) + } + + /// Mark a credential unusable, in the filesystem and in memory. + pub async fn revoke(&self, credential_id: &[u8]) -> Result<()> { + let mut stored = self + .read(credential_id) + .await? + .context("no such credential")?; + stored.active = false; + self.overwrite(credential_id, &stored).await?; + // The cache is what the gate reads, so it must not outlive the + // decision to revoke by even one request. + if let Some(cached) = self + .cache + .lock() + .expect("credentials poisoned") + .get_mut(credential_id) + { + cached.active = false; + } + Ok(()) + } + + async fn overwrite(&self, credential_id: &[u8], record: &StoredCredential) -> Result<()> { + let path = format!( + "/{RUNTIME_DIR}/{CREDENTIALS_DIR}/{}", + Self::name(credential_id) + ); + let mut bytes = Vec::new(); + ciborium::into_writer(record, &mut bytes).context("encoding a credential record")?; + + let handle = self + .fs + .open(&path, OpenFlags::write_only()) + .await + .map_err(|e| anyhow::anyhow!(e)) + .with_context(|| format!("opening {path}"))?; + self.fs + .set_size(&handle, 0) + .await + .map_err(|e| anyhow::anyhow!(e)) + .context("truncating a credential record")?; + self.fs + .pwrite(&handle, 0, &bytes) + .await + .map_err(|e| anyhow::anyhow!(e)) + .context("writing a credential record")?; + self.fs + .sync(&handle) + .await + .map_err(|e| anyhow::anyhow!(e)) + .context("committing a credential record")?; + self.fs + .close(&handle) + .await + .map_err(|e| anyhow::anyhow!(e)) + .context("closing a credential record")?; + Ok(()) + } + + async fn read(&self, credential_id: &[u8]) -> Result> { + let path = format!( + "/{RUNTIME_DIR}/{CREDENTIALS_DIR}/{}", + Self::name(credential_id) + ); + let handle = match self.fs.open(&path, OpenFlags::read_only()).await { + Ok(handle) => handle, + Err(FsError::NotFound) => return Ok(None), + Err(e) => return Err(anyhow::anyhow!(e)).with_context(|| format!("opening {path}")), + }; + let bytes = self.fs.pread(&handle, 0, MAX_RECORD_BYTES).await; + // Before `?`, so a read error does not leak the handle on every attempt. + self.fs + .close(&handle) + .await + .map_err(|e| anyhow::anyhow!(e)) + .context("closing a credential record")?; + let bytes = bytes + .map_err(|e| anyhow::anyhow!(e)) + .context("reading a credential record")?; + + let stored: StoredCredential = + ciborium::from_reader(bytes.as_ref()).context("decoding a credential record")?; + anyhow::ensure!( + stored.version == RECORD_VERSION, + "credential record is version {}, this build understands {RECORD_VERSION}", + stored.version + ); + Ok(Some(stored)) + } +} + +#[async_trait::async_trait] +impl CredentialStore for FilesystemCredentials { + async fn lookup(&self, credential_id: &[u8]) -> Result> { + if let Some(hit) = self + .cache + .lock() + .expect("credentials poisoned") + .get(credential_id) + { + return Ok(Some(hit.clone())); + } + let Some(stored) = self.read(credential_id).await? else { + return Ok(None); + }; + let record = CredentialRecord { + tenant_id: stored.tenant_id, + passkey: stored.passkey, + active: stored.active, + counter: stored.counter, + }; + self.cache + .lock() + .expect("credentials poisoned") + .insert(credential_id.to_vec(), record.clone()); + Ok(Some(record)) + } + + async fn record_use(&self, credential_id: &[u8], counter: u32) -> Result<()> { + let mut cache = self.cache.lock().expect("credentials poisoned"); + let Some(record) = cache.get_mut(credential_id) else { + return Ok(()); + }; + // Zero means the authenticator does not keep one — which is what a + // synced platform passkey reports — so it says nothing either way and + // is not a regression. + if counter != 0 && counter <= record.counter { + tracing::warn!( + credential = %hex::encode(&credential_id[..8.min(credential_id.len())]), + stored = record.counter, + presented = counter, + "authenticator sign counter did not advance; this can mean a cloned \ + credential and is recorded rather than acted on, because the policy for \ + it has to be a deliberate decision" + ); + } + record.counter = record.counter.max(counter); + Ok(()) + } +} + +/// Thirty-two bytes of tenant identity, from the enclave's own entropy. +/// +/// The NSM rather than the host's RNG: an identifier the host could predict +/// would let it create a tenant directory before the tenant did. +pub fn mint_tenant_id(entropy: &Arc) -> Result<[u8; 16]> { + let mut id = [0u8; 16]; + entropy + .get_random(&mut id) + .context("drawing a tenant id from the NSM")?; + Ok(id) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::auth::testing::{Relying, RP_ID}; + use crate::auth::SoftwareAuthenticator; + use s3fs_core::backend::memory::MemoryBackend; + use s3fs_core::{Config, MasterSecret}; + + async fn shared_fs() -> Arc { + let backend = Arc::new(MemoryBackend::new()); + Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([7u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("the shared filesystem") + } + + fn stored(tenant_id: [u8; 16], passkey: Passkey) -> StoredCredential { + StoredCredential { + version: RECORD_VERSION, + tenant_id, + passkey, + active: true, + created_ms: 0, + counter: 0, + } + } + + async fn registered() -> ( + Arc, + FilesystemCredentials, + SoftwareAuthenticator, + [u8; 16], + ) { + let fs = shared_fs().await; + let store = FilesystemCredentials::new(fs.clone()); + let rp = Relying::new(); + let auth = SoftwareAuthenticator::new(RP_ID); + let passkey = rp.register(&auth); + let tenant_id = [0x11; 16]; + store + .register(auth.credential_id(), stored(tenant_id, passkey)) + .await + .expect("registering"); + (fs, store, auth, tenant_id) + } + + /// A record written on one boot is readable on the next: the point of + /// putting it in the anchored filesystem rather than in memory. + #[tokio::test] + async fn a_credential_survives_a_restart() { + let (fs, _store, auth, tenant_id) = registered().await; + + // A second store over the same filesystem, with an empty cache — as + // close to a restart as this gets without a second process. + let after = FilesystemCredentials::new(fs); + let found = after + .lookup(auth.credential_id()) + .await + .unwrap() + .expect("the credential is still there"); + assert_eq!(found.tenant_id, tenant_id); + assert!(found.active); + } + + #[tokio::test] + async fn an_unregistered_credential_is_absent_rather_than_an_error() { + let fs = shared_fs().await; + let store = FilesystemCredentials::new(fs); + assert!(store.lookup(&[0xff; 32]).await.unwrap().is_none()); + } + + /// Registering the same credential twice must not silently repoint it — + /// that is what taking over somebody else's tenant would look like. + #[tokio::test] + async fn a_credential_cannot_be_registered_over() { + let (_fs, store, auth, _) = registered().await; + let rp = Relying::new(); + let other = SoftwareAuthenticator::new(RP_ID); + let passkey = rp.register(&other); + assert!(store + .register(auth.credential_id(), stored([0x22; 16], passkey)) + .await + .is_err()); + } + + /// Revocation must be visible immediately — the gate reads the cache, and + /// a cache that outlived the decision by even one request would be one + /// more signature from a lost phone. + #[tokio::test] + async fn revocation_takes_effect_at_once_and_survives_a_restart() { + let (fs, store, auth, _) = registered().await; + assert!( + store + .lookup(auth.credential_id()) + .await + .unwrap() + .unwrap() + .active + ); + + store.revoke(auth.credential_id()).await.unwrap(); + assert!( + !store + .lookup(auth.credential_id()) + .await + .unwrap() + .unwrap() + .active + ); + + let after = FilesystemCredentials::new(fs); + assert!( + !after + .lookup(auth.credential_id()) + .await + .unwrap() + .unwrap() + .active + ); + } + + /// Several passkeys, one tenant. The reason the tenant id is minted rather + /// than derived from a credential. + #[tokio::test] + async fn many_credentials_can_share_one_tenant() { + let fs = shared_fs().await; + let store = FilesystemCredentials::new(fs); + let rp = Relying::new(); + let tenant_id = [0x33; 16]; + + let devices: Vec<_> = (0..3).map(|_| SoftwareAuthenticator::new(RP_ID)).collect(); + for device in &devices { + let passkey = rp.register(device); + store + .register(device.credential_id(), stored(tenant_id, passkey)) + .await + .unwrap(); + } + for device in &devices { + let found = store.lookup(device.credential_id()).await.unwrap().unwrap(); + assert_eq!( + found.tenant_id, tenant_id, + "a device reached another tenant" + ); + } + + // And revoking one leaves the others working, which is the whole point + // of being able to lose a phone. + store.revoke(devices[0].credential_id()).await.unwrap(); + assert!( + !store + .lookup(devices[0].credential_id()) + .await + .unwrap() + .unwrap() + .active + ); + assert!( + store + .lookup(devices[1].credential_id()) + .await + .unwrap() + .unwrap() + .active + ); + } + + /// A counter that does not advance is recorded, not fatal — a synced + /// passkey reports zero forever and must keep working. + #[tokio::test] + async fn a_zero_counter_is_not_treated_as_a_regression() { + let (_fs, store, auth, _) = registered().await; + store.record_use(auth.credential_id(), 5).await.unwrap(); + store.record_use(auth.credential_id(), 0).await.unwrap(); + // Never moves backwards, and the credential stays usable. + assert_eq!( + store + .lookup(auth.credential_id()) + .await + .unwrap() + .unwrap() + .counter, + 5 + ); + assert!( + store + .lookup(auth.credential_id()) + .await + .unwrap() + .unwrap() + .active + ); + } + + /// Records are runtime-owned and sit above every tenant scope, so a guest + /// cannot name them. This asserts the location, which is what the resolver + /// guarantee rests on. + #[tokio::test] + async fn credentials_live_above_every_tenant_scope() { + let (fs, _store, auth, _) = registered().await; + let root = fs.root(); + let runtime = fs.lookup_at(&root, RUNTIME_DIR).await.unwrap(); + let creds = fs.lookup_at(&runtime, CREDENTIALS_DIR).await.unwrap(); + assert!(fs + .lookup_at(&creds, &hex::encode(auth.credential_id())) + .await + .is_ok()); + + // A tenant's scope is a sibling of `/runtime`, never an ancestor. + let tenant = crate::tenant::tenant_root_by_id(&fs, [0x11; 16]) + .await + .unwrap(); + assert_ne!(tenant.scope.objid(), root.objid()); + assert_ne!(tenant.scope.objid(), runtime.objid()); + } +} diff --git a/runtime/src/auth/gate.rs b/runtime/src/auth/gate.rs new file mode 100644 index 0000000..f77da66 --- /dev/null +++ b/runtime/src/auth/gate.rs @@ -0,0 +1,623 @@ +//! The gate: no verified assertion, no guest. +//! +//! Everything a request must survive before a tenant exists, let alone a warm +//! instance. [`Gate::verify`] is the only way past it and it either returns a +//! [`Verified`] tenant or a [`Denied`]; there is no third outcome and no +//! caller-supplied way to skip it. +//! +//! ## What is checked where +//! +//! `webauthn-rs` owns the assertion itself — `clientDataJSON` names +//! `webauthn.get` and the challenge that was issued, the origin matches +//! exactly, `rpIdHash` is this relying party, user *verification* happened +//! rather than mere presence, and the signature verifies under the stored key. +//! +//! This module owns everything the protocol has no field for: +//! +//! - the challenge exists, has not expired, and has not been used before; +//! - the credential is one this runtime knows and has not revoked; +//! - and the request that arrived is the request the challenge was issued for. +//! +//! That last one is the reason any of this is worth doing. Without it an +//! approval the user gave for one transaction would authorize a different one, +//! and a cosigner that can be made to sign a substituted payload is not a +//! cosigner. +//! +//! ## Why the body is buffered here +//! +//! The binding commits to `sha256(body)`, so the runtime has to have the whole +//! body before it can check it. It is read to a bounded buffer and the hash is +//! computed from **the bytes that will actually be forwarded** — never from a +//! length or digest the client supplied, which would let the client decide what +//! it was approving. + +use std::sync::Arc; + +use webauthn_rs::prelude::*; + +use super::challenge::{ChallengeError, ChallengeStore}; +use super::token::{InteractionScope, TokenError, TokenStore}; + +/// Names the challenge a request is answering. +pub const CHALLENGE_HEADER: &str = "x-webauthn-challenge-id"; +/// Carries the assertion: base64url of the `PublicKeyCredential` JSON that +/// `navigator.credentials.get()` produced. +/// +/// One header holding the credential verbatim, rather than five holding its +/// parts. The client already has this JSON; splitting it up would mean +/// reassembling a structure by hand on the verifying side, which is exactly +/// where a subtle mismatch would hide. +pub const ASSERTION_HEADER: &str = "x-webauthn-assertion"; + +/// Carries the interaction token: `Authorization: Bearer `. +/// +/// An ordinary HTTP header rather than an `x-enclave-*` one, deliberately: a +/// gRPC stream's opening metadata *is* an HTTP/2 HEADERS frame, so one header +/// covers a request and a stream alike, and every client library already knows +/// how to set this one. +pub const AUTHORIZATION_HEADER: &str = "authorization"; + +/// Every header the gate consumes. Stripped before the guest sees a request, so +/// a guest can neither read a client's token nor forge one. +pub const AUTH_HEADERS: [&str; 1] = [AUTHORIZATION_HEADER]; + +/// Ceiling on a bearer token, in characters. +/// +/// The runtime mints 32 bytes as base64url, so anything much longer is not one +/// of ours; the bound exists so a header cannot make the runtime hash an +/// arbitrary amount of input before deciding it was never valid. +const MAX_TOKEN_CHARS: usize = 128; + +/// What a client is known as, once it has proved it. +#[derive(Debug, Clone)] +pub struct Verified { + /// Sixteen bytes, matching [`crate::tenant::TenantRoot`] — the plan said + /// thirty-two, but the directory layout already used sixteen and one + /// number for one thing is worth more than the extra bits. A minted, + /// collision-checked 128-bit identifier is ample; it is not a secret and + /// nothing derives a key from it. + pub tenant_id: [u8; 16], +} + +/// What a passkey ceremony established, before a token was minted for it. +#[derive(Debug, Clone)] +pub struct Authenticated { + pub tenant_id: [u8; 16], + pub credential_id: Vec, + /// The interaction the challenge was issued for, which is what the token + /// will be good for and nothing else. + pub scope: InteractionScope, +} + +/// Why a request did not get past the gate. +/// +/// The variants exist for the runtime's log. **Clients are told one thing.** An +/// attacker distinguishing "unknown credential" from "bad signature" learns +/// which guesses to keep making; a legitimate client learns nothing it can act +/// on, because the remedy for all of them is to fetch a new challenge. +#[derive(Debug)] +pub enum Denied { + /// No assertion at all. + Missing, + /// Present but not parseable, or over the size limit. + Malformed(String), + Challenge(ChallengeError), + Token(TokenError), + /// The request is not the one the challenge was issued for. + NotTheRequest, + /// No such credential, or it has been revoked. + UnknownCredential, + /// The assertion did not verify. + Assertion(String), + /// The body was larger than the runtime will buffer. + BodyTooLarge, + /// Reading the body failed. + Body(String), +} + +impl Denied { + /// What the client is told. Deliberately the same for every variant. + pub fn public_message(&self) -> &'static str { + match self { + Denied::BodyTooLarge => "request body too large", + _ => "a fresh WebAuthn assertion bound to this request is required", + } + } + + pub fn status(&self) -> hyper::StatusCode { + match self { + Denied::BodyTooLarge => hyper::StatusCode::PAYLOAD_TOO_LARGE, + _ => hyper::StatusCode::UNAUTHORIZED, + } + } +} + +impl std::fmt::Display for Denied { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Denied::Missing => write!(f, "no assertion presented"), + Denied::Malformed(e) => write!(f, "malformed assertion: {e}"), + Denied::Challenge(e) => write!(f, "{e}"), + Denied::Token(e) => write!(f, "{e}"), + Denied::NotTheRequest => { + write!(f, "the assertion was issued for a different request") + } + Denied::UnknownCredential => write!(f, "unknown or revoked credential"), + Denied::Assertion(e) => write!(f, "assertion did not verify: {e}"), + Denied::BodyTooLarge => write!(f, "request body over the limit"), + Denied::Body(e) => write!(f, "reading the request body: {e}"), + } + } +} + +/// One registered passkey, and the tenant it speaks for. +#[derive(Debug, Clone)] +pub struct CredentialRecord { + pub tenant_id: [u8; 16], + pub passkey: Passkey, + /// A revoked credential stays on record rather than being deleted, so a + /// lost phone's key can be refused by name rather than merely forgotten. + pub active: bool, + pub counter: u32, +} + +/// Where registered credentials live. +/// +/// A trait so the gate can be tested without a filesystem, and so the +/// filesystem-backed implementation can arrive separately without the gate +/// changing. +#[async_trait::async_trait] +pub trait CredentialStore: Send + Sync { + async fn lookup(&self, credential_id: &[u8]) -> anyhow::Result>; + /// Record that this credential was used, and at what counter. + async fn record_use(&self, credential_id: &[u8], counter: u32) -> anyhow::Result<()>; +} + +impl std::fmt::Debug for Gate { + /// Names what it is configured for, never what it holds. A `Debug` that + /// printed challenges would put single-use secrets in any log line that + /// formatted a struct containing one. + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("Gate") + .field("outstanding_challenges", &self.challenges.outstanding()) + .field("outstanding_tokens", &self.tokens.outstanding()) + .finish_non_exhaustive() + } +} + +pub struct Gate { + webauthn: Webauthn, + challenges: ChallengeStore, + credentials: Arc, + tokens: TokenStore, +} + +impl Gate { + pub fn new( + webauthn: Webauthn, + challenges: ChallengeStore, + credentials: Arc, + tokens: TokenStore, + ) -> Self { + Gate { + webauthn, + challenges, + credentials, + tokens, + } + } + + pub fn webauthn(&self) -> &Webauthn { + &self.webauthn + } + + pub fn challenges(&self) -> &ChallengeStore { + &self.challenges + } + + pub fn credentials(&self) -> &Arc { + &self.credentials + } + + pub fn tokens(&self) -> &TokenStore { + &self.tokens + } + + /// Issue a challenge for one intended interaction. + /// + /// The scope is recorded now and compared when the token it leads to is + /// spent; the client cannot influence it after the fact because it never + /// sees the record. + pub fn issue( + &self, + id: [u8; 16], + scope: InteractionScope, + allowed: &[Passkey], + ) -> Result { + let (options, state) = self + .webauthn + .start_passkey_authentication(allowed) + .map_err(|e| Denied::Assertion(e.to_string()))?; + self.challenges + .issue(id, scope, state) + .map_err(Denied::Challenge)?; + Ok(options) + } + + /// The passkey half: prove who is asking, and what they are asking for. + /// + /// This is everything the old per-request gate did except reading a body, + /// in the same order and with the same refusals. What it does *not* do is + /// let anything through — it establishes an identity and an interaction, and + /// the caller mints a token for that pair. The interaction itself presents + /// the token. + pub async fn authenticate( + &self, + challenge_id: &[u8; 16], + assertion: &PublicKeyCredential, + ) -> Result { + // Consumed whatever happens next: an assertion that was offered and + // refused has still been seen by whoever sent it, and a second attempt + // at the same challenge buys them only attempts. + let (issued_for, state) = self + .challenges + .consume(challenge_id) + .map_err(Denied::Challenge)?; + + let record = self + .credentials + .lookup(assertion.raw_id.as_ref()) + .await + .map_err(|e| Denied::Assertion(e.to_string()))? + .filter(|r| r.active) + .ok_or(Denied::UnknownCredential)?; + + let result = self + .webauthn + .finish_passkey_authentication(assertion, &state) + .map_err(|e| Denied::Assertion(e.to_string()))?; + + // Belt and braces: `start_passkey_authentication` sets + // `UserVerificationPolicy::Required`, so the crate has already refused + // an unverified assertion. Checked again because "a person did this" + // is the entire claim a cosigner rests on, and a policy that moved + // upstream should not silently weaken this. + if !result.user_verified() { + return Err(Denied::Assertion("user was not verified".into())); + } + + self.credentials + .record_use(assertion.raw_id.as_ref(), result.counter()) + .await + .map_err(|e| Denied::Assertion(e.to_string()))?; + + Ok(Authenticated { + tenant_id: record.tenant_id, + credential_id: assertion.raw_id.as_ref().to_vec(), + scope: issued_for, + }) + } + + /// Record an approval as a token, and hand back the token's bytes. + /// + /// Generated here rather than by the caller so that the only copy outside + /// this function is the one going to the client — the store keeps a hash. + pub fn grant(&self, token: &[u8], who: &Authenticated) -> Result { + self.tokens + .issue(token, who.tenant_id, who.scope.clone()) + .map_err(Denied::Token)?; + Ok(self.tokens.ttl()) + } + + /// The token half: spend one approval on one interaction. + /// + /// The scope is rebuilt from the request that actually arrived and compared + /// with the one the person approved, so a token cannot be moved to another + /// route. It says nothing about the body, and deliberately: an interaction + /// may be a stream, whose body does not exist when the approval is given. + pub fn redeem(&self, req: &hyper::Request) -> Result { + let offered = req + .headers() + .get(AUTHORIZATION_HEADER) + .and_then(|v| v.to_str().ok()) + .and_then(|v| v.strip_prefix("Bearer ")) + .map(str::trim) + .ok_or(Denied::Missing)?; + if offered.is_empty() || offered.len() > MAX_TOKEN_CHARS { + return Err(Denied::Malformed("token is not a plausible length".into())); + } + + let arrived_as = + InteractionScope::new(req.method().as_str(), req.uri().path(), req.uri().query()); + let tenant_id = self + .tokens + .redeem(offered.as_bytes(), &arrived_as) + .map_err(Denied::Token)?; + + Ok(Verified { tenant_id }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::auth::authenticator::flags; + use crate::auth::testing::{Harness, Relying, ORIGIN}; + use bytes::Bytes; + + fn credential(assertion: &serde_json::Value) -> PublicKeyCredential { + serde_json::from_str(&assertion.to_string()).expect("a credential") + } + + /// Issue a challenge for `scope` and answer it with `authenticator`. + async fn answer( + h: &Harness, + id: [u8; 16], + scope: InteractionScope, + authenticator: &crate::auth::SoftwareAuthenticator, + origin: &str, + ) -> Result { + let options = h + .gate + .issue(id, scope, std::slice::from_ref(&h.passkey)) + .expect("issuing a challenge"); + let assertion = authenticator.assert(&Relying::challenge_for(&options), origin); + h.gate.authenticate(&id, &credential(&assertion)).await + } + + fn scope() -> InteractionScope { + InteractionScope::new("POST", "/sign", None) + } + + // --- the passkey half ------------------------------------------------- + + /// The happy path, so every refusal below means something. + #[tokio::test] + async fn an_assertion_identifies_the_tenant_and_the_interaction() { + let h = Harness::new(); + let who = answer(&h, [1u8; 16], scope(), &h.authenticator, ORIGIN) + .await + .expect("verifies"); + assert_eq!(who.tenant_id, h.tenant_id); + assert_eq!(who.credential_id, h.authenticator.credential_id()); + assert_eq!( + who.scope, + scope(), + "the approval named a different interaction" + ); + } + + /// **The replay.** One challenge, one assertion, once. + #[tokio::test] + async fn a_challenge_cannot_be_answered_twice() { + let h = Harness::new(); + let id = [2u8; 16]; + let options = h + .gate + .issue(id, scope(), std::slice::from_ref(&h.passkey)) + .unwrap(); + let assertion = credential( + &h.authenticator + .assert(&Relying::challenge_for(&options), ORIGIN), + ); + h.gate.authenticate(&id, &assertion).await.expect("first"); + assert!(matches!( + h.gate.authenticate(&id, &assertion).await, + Err(Denied::Challenge(ChallengeError::Unknown)) + )); + } + + #[tokio::test] + async fn a_revoked_credential_is_refused() { + let h = Harness::new(); + h.credentials.revoke(h.authenticator.credential_id()); + assert!(matches!( + answer(&h, [3u8; 16], scope(), &h.authenticator, ORIGIN).await, + Err(Denied::UnknownCredential) + )); + } + + /// A well-formed assertion from a passkey this runtime never registered. + #[tokio::test] + async fn an_unregistered_credential_is_refused() { + let h = Harness::new(); + let stranger = crate::auth::SoftwareAuthenticator::new("enclave.test"); + assert!(matches!( + answer(&h, [4u8; 16], scope(), &stranger, ORIGIN).await, + Err(Denied::UnknownCredential) + )); + } + + /// A phone in a pocket is not a person approving anything. + #[tokio::test] + async fn an_assertion_from_another_origin_is_refused() { + let h = Harness::new(); + assert!(matches!( + answer( + &h, + [5u8; 16], + scope(), + &h.authenticator, + "https://attacker.example" + ) + .await, + Err(Denied::Assertion(_)) + )); + } + + /// **"A person did this" is the entire claim.** Mere presence is not it. + #[tokio::test] + async fn an_assertion_without_user_verification_is_refused() { + let h = Harness::new(); + let id = [6u8; 16]; + let options = h + .gate + .issue(id, scope(), std::slice::from_ref(&h.passkey)) + .unwrap(); + let assertion = + h.authenticator + .assert_with(&Relying::challenge_for(&options), ORIGIN, flags::UP); + assert!(matches!( + h.gate.authenticate(&id, &credential(&assertion)).await, + Err(Denied::Assertion(_)) + )); + } + + #[tokio::test] + async fn a_challenge_that_was_never_issued_is_refused() { + let h = Harness::new(); + let id = [7u8; 16]; + let options = h + .gate + .issue(id, scope(), std::slice::from_ref(&h.passkey)) + .unwrap(); + let assertion = credential( + &h.authenticator + .assert(&Relying::challenge_for(&options), ORIGIN), + ); + assert!(matches!( + h.gate.authenticate(&[0xff; 16], &assertion).await, + Err(Denied::Challenge(ChallengeError::Unknown)) + )); + } + + /// **Under concurrency, one winner.** Two threads answering one challenge. + #[tokio::test(flavor = "multi_thread")] + async fn two_simultaneous_answers_to_one_challenge_race_to_exactly_one_winner() { + for _ in 0..25 { + let h = std::sync::Arc::new(Harness::new()); + let id = [8u8; 16]; + let options = h + .gate + .issue(id, scope(), std::slice::from_ref(&h.passkey)) + .unwrap(); + let assertion = credential( + &h.authenticator + .assert(&Relying::challenge_for(&options), ORIGIN), + ); + + let (a, b) = { + let (h1, h2) = (h.clone(), h.clone()); + let (c1, c2) = (assertion.clone(), assertion.clone()); + tokio::join!( + tokio::spawn(async move { h1.gate.authenticate(&id, &c1).await.is_ok() }), + tokio::spawn(async move { h2.gate.authenticate(&id, &c2).await.is_ok() }), + ) + }; + let winners = usize::from(a.unwrap()) + usize::from(b.unwrap()); + assert_eq!(winners, 1, "one challenge was answered {winners} times"); + } + } + + // --- the token half --------------------------------------------------- + + /// A token identifies its tenant, and spends itself doing it. + #[tokio::test] + async fn a_token_admits_one_interaction() { + let h = Harness::new(); + let token = h.token_for("POST", "/sign").await; + let req = Harness::bearer("POST", "/sign", b"pay alice", &token); + assert_eq!(h.gate.redeem(&req).unwrap().tenant_id, h.tenant_id); + + let again = Harness::bearer("POST", "/sign", b"pay alice", &token); + assert!( + matches!(h.gate.redeem(&again), Err(Denied::Token(_))), + "a token admitted a second interaction" + ); + } + + /// **What the token still refuses.** It names a route, and cannot be moved. + #[tokio::test] + async fn a_token_cannot_be_moved_to_another_interaction() { + for (method, path) in [ + ("POST", "/withdraw"), + ("GET", "/sign"), + ("POST", "/sign?all=1"), + ] { + let h = Harness::new(); + let token = h.token_for("POST", "/sign").await; + let req = Harness::bearer(method, path, b"", &token); + assert!( + matches!(h.gate.redeem(&req), Err(Denied::Token(_))), + "an approval for POST /sign was spent on {method} {path}" + ); + } + } + + /// **What it deliberately does not refuse, and this is the trade.** + /// + /// The approval names the interaction, not the payload. A different body at + /// the same route is the same interaction as far as the runtime is + /// concerned, and the person who approved it saw no bytes. This test exists + /// so the property is written down rather than discovered. + #[tokio::test] + async fn a_token_does_not_bind_the_body() { + let h = Harness::new(); + let token = h.token_for("POST", "/sign").await; + let req = Harness::bearer("POST", "/sign", b"pay mallory 1000", &token); + assert!( + h.gate.redeem(&req).is_ok(), + "the token bound a body it was never given" + ); + } + + /// The whole rule: nothing gets through without one. + #[tokio::test] + async fn a_request_without_a_token_is_refused() { + let h = Harness::new(); + let req = hyper::Request::builder() + .method("POST") + .uri("https://enclave.test/sign") + .body(http_body_util::Full::new(Bytes::new())) + .expect("well-formed request"); + assert!(matches!(h.gate.redeem(&req), Err(Denied::Missing))); + } + + #[tokio::test] + async fn a_malformed_authorization_header_is_refused_rather_than_panicking() { + let h = Harness::new(); + for value in [ + "", + "Bearer ", + "Basic abc", + "bearer lowercase-scheme", + &"x".repeat(4096), + ] { + let req = hyper::Request::builder() + .method("POST") + .uri("https://enclave.test/sign") + .header(AUTHORIZATION_HEADER, value) + .body(http_body_util::Full::new(Bytes::new())) + .expect("well-formed request"); + assert!( + h.gate.redeem(&req).is_err(), + "{value:?} was accepted as a token" + ); + } + } + + /// **The uniformity guarantee.** Every refusal says the same thing. + #[test] + fn refusals_are_indistinguishable_to_the_client() { + let all = [ + Denied::Missing, + Denied::Malformed("x".into()), + Denied::Challenge(ChallengeError::Unknown), + Denied::Token(crate::auth::TokenError::WrongInteraction), + Denied::Token(crate::auth::TokenError::Expired), + Denied::UnknownCredential, + Denied::Assertion("x".into()), + ]; + for denied in &all { + assert_eq!( + denied.public_message(), + all[0].public_message(), + "{denied} tells a client something the others do not" + ); + let public = denied.public_message(); + assert!(!public.contains("credential"), "{public}"); + assert!(!public.contains("signature"), "{public}"); + assert!(!public.contains("expired"), "{public}"); + } + } +} diff --git a/runtime/src/auth/mod.rs b/runtime/src/auth/mod.rs new file mode 100644 index 0000000..f5440f8 --- /dev/null +++ b/runtime/src/auth/mod.rs @@ -0,0 +1,174 @@ +//! Who is allowed to reach the guest, and for which request. +//! +//! **No valid, fresh, single-use WebAuthn assertion bound to this exact +//! request means the guest is never called.** No session, no cookie, no bearer +//! token and no certificate authorizes a guest request by itself. +//! +//! ## The division of labour +//! +//! `webauthn-rs` verifies the assertion: that `clientDataJSON` names +//! `webauthn.get` and the challenge it was issued, that the origin matches +//! exactly, that `rpIdHash` is this relying party, that user *verification* +//! happened rather than mere presence, and that the signature checks out under +//! the credential's stored key. +//! +//! What it cannot know is which HTTP request any of that was for — WebAuthn has +//! no field for it. So [`challenge`] records, when a challenge is issued, the +//! method, path, query and body hash it was issued for, and the gate compares +//! that record against the request that actually arrived. Without it an +//! approval for one transaction would authorize another. +//! +//! ## What a signature does and does not prove +//! +//! An authenticator displays nothing. The user approves *a prompt at a moment*, +//! not a payload they read. So "the user approved this exact operation" is +//! precise only because the runtime chose the challenge, remembered what it was +//! for, and will accept it for nothing else — the strength is in the binding, +//! not in the ceremony. + +/// A passkey in software, for tests and for the QEMU harness. +/// +/// Behind a feature because it *forges assertions*. Nothing that can mint an +/// approval should be reachable in a production image, however carefully it is +/// otherwise unused. +#[cfg(any(test, feature = "testing"))] +pub mod authenticator; +pub mod challenge; +pub mod credential; +pub mod gate; +pub mod routes; +#[cfg(any(test, feature = "testing"))] +pub mod testing; +mod token; + +#[cfg(any(test, feature = "testing"))] +pub use authenticator::SoftwareAuthenticator; +pub use challenge::{ChallengeError, ChallengeStore, DEFAULT_CAPACITY, DEFAULT_TTL}; +pub use credential::{mint_tenant_id, FilesystemCredentials, StoredCredential}; +pub use gate::{ + Authenticated, CredentialRecord, CredentialStore, Denied, Gate, Verified, AUTHORIZATION_HEADER, + AUTH_HEADERS, +}; +pub use routes::{AuthEndpoints, AUTH_PREFIX}; +pub use token::{ + InteractionScope, TokenError, TokenStore, DEFAULT_CAPACITY as DEFAULT_TOKEN_CAPACITY, + DEFAULT_TTL as DEFAULT_TOKEN_TTL, +}; + +/// Build the relying party from a domain, an origin, and any further origins +/// assertions may claim. +/// +/// All are checked here rather than deep inside a request: an enclave that +/// booted with a malformed origin would otherwise refuse every assertion at +/// runtime, with the reason buried. +/// +/// `allowed_origins` exists for native apps, which do not claim +/// `https://`. An Android app using Credential Manager claims +/// `android:apk-key-hash:`, and Android only lets it claim that after +/// checking the app against `https:///.well-known/assetlinks.json` — so +/// the origin names an app the domain vouched for, not a page anyone can serve. +/// Every origin is still compared exactly. +pub fn build_relying_party( + rp_id: &str, + origin: &str, + allowed_origins: &[String], +) -> anyhow::Result { + use anyhow::Context as _; + let url = webauthn_rs::prelude::Url::parse(origin) + .with_context(|| format!("--webauthn-origin {origin:?} is not a URL"))?; + let mut builder = webauthn_rs::WebauthnBuilder::new(rp_id, &url) + .with_context(|| format!("relying party {rp_id:?} at {origin:?}"))?; + for allowed in allowed_origins { + builder = builder.append_allowed_origin(&parse_allowed_origin(allowed)?); + } + builder.build().context("building the relying party") +} + +/// Parse one `--webauthn-allowed-origin`, refusing the mistakes that would +/// otherwise surface only as every assertion from the app being refused. +/// +/// An Android origin is checked for shape: the hash is the unpadded base64url +/// SHA-256 of the signing certificate. The likeliest wrong value is the +/// colon-separated hex fingerprint `keytool` and the Play Console print, which +/// names the same certificate but can never equal what Android sends. +fn parse_allowed_origin(origin: &str) -> anyhow::Result { + use anyhow::Context as _; + use base64::Engine as _; + + let url = webauthn_rs::prelude::Url::parse(origin) + .with_context(|| format!("--webauthn-allowed-origin {origin:?} is not a URL"))?; + if url.scheme() == "android" { + let hash = url.path().strip_prefix("apk-key-hash:").with_context(|| { + format!( + "--webauthn-allowed-origin {origin:?}: an Android origin has the form \ + android:apk-key-hash:" + ) + })?; + let digest = base64::engine::general_purpose::URL_SAFE_NO_PAD + .decode(hash) + .ok() + .filter(|d| d.len() == 32); + anyhow::ensure!( + digest.is_some(), + "--webauthn-allowed-origin {origin:?}: the hash must be the SHA-256 of the \ + app's signing certificate as unpadded base64url (43 characters). A \ + colon-separated hex fingerprint from keytool or the Play Console names the \ + same certificate but must be converted: Android never sends it in that form." + ); + } + Ok(url) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// SHA-256 of the empty string, in the form Android sends. + const APK_KEY_HASH: &str = "android:apk-key-hash:47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU"; + + #[test] + fn an_android_origin_is_accepted() { + assert_eq!( + parse_allowed_origin(APK_KEY_HASH).unwrap().as_str(), + APK_KEY_HASH + ); + build_relying_party( + "enclave.test", + "https://enclave.test", + &[APK_KEY_HASH.into()], + ) + .expect("relying party"); + } + + #[test] + fn a_hex_fingerprint_is_refused_with_the_reason() { + let err = parse_allowed_origin( + "android:apk-key-hash:E3:B0:C4:42:98:FC:1C:14:9A:FB:F4:C8:99:6F:B9:24:27:AE:41:E4:64:9B:93:4C:A4:95:99:1B:78:52:B8:55", + ) + .unwrap_err(); + assert!(format!("{err:#}").contains("hex fingerprint"), "{err:#}"); + } + + #[test] + fn a_truncated_or_padded_hash_is_refused() { + assert!(parse_allowed_origin(&APK_KEY_HASH[..APK_KEY_HASH.len() - 1]).is_err()); + assert!(parse_allowed_origin(&format!("{APK_KEY_HASH}=")).is_err()); + } + + #[test] + fn an_android_origin_without_the_key_hash_prefix_is_refused() { + assert!( + parse_allowed_origin("android:47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU").is_err() + ); + } + + #[test] + fn something_that_is_not_a_url_is_refused() { + assert!(build_relying_party( + "enclave.test", + "https://enclave.test", + &["not a url".into()] + ) + .is_err()); + } +} diff --git a/runtime/src/auth/routes.rs b/runtime/src/auth/routes.rs new file mode 100644 index 0000000..6d43e6a --- /dev/null +++ b/runtime/src/auth/routes.rs @@ -0,0 +1,813 @@ +//! `/auth/*` — the only routes that answer without an assertion. +//! +//! They are runtime-owned in the same sense `/enclave/*` is: refused before +//! dispatch, never forwarded, and unreachable by any guest. That is not a +//! convenience. A guest able to answer under `/auth/` could hand out its own +//! challenges and verify its own assertions, which is the same as having none. +//! +//! Nothing here performs a cosigner action or touches an *existing* tenant's +//! directory. The most any of it does is create a new tenant, which anyone may +//! do: registration is open, and what it grants is an empty tenant and nothing +//! else. +//! +//! ## Why registration is two calls +//! +//! WebAuthn registration is a challenge and a response, and the runtime has to +//! remember which challenge it issued while the user holds their thumb to a +//! phone. The state is kept here, keyed by an id the client quotes back, for +//! the same reasons the assertion challenges are: single use, short lived, and +//! bounded, because issuing one is necessarily unauthenticated. + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; +use std::time::{Duration, Instant}; + +use base64::Engine as _; +use bytes::Bytes; +use serde::{Deserialize, Serialize}; +use webauthn_rs::prelude::*; + +use super::credential::{FilesystemCredentials, StoredCredential}; +use super::gate::Gate; +use super::token::InteractionScope; + +/// The prefix the guest can never see. +pub const AUTH_PREFIX: &str = "/auth/"; + +/// How long a half-finished registration is held. +const REGISTRATION_TTL: Duration = Duration::from_secs(300); +/// Half-finished registrations allowed at once. +const MAX_REGISTRATIONS: usize = 64; +/// Ceiling on an `/auth/*` request body. +const MAX_BODY: usize = 16 * 1024; + +/// Registration is open, so nothing here is required. The only thing a caller +/// may supply is what to call the credential. +/// +/// `deny_unknown_fields` for the same reason `RequestOptionsRequest` has it: an +/// older client still sending `enrollment_token` is told that it means nothing +/// now, rather than having it quietly ignored and believing it was admitted on +/// the strength of one. +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +struct RegisterOptionsRequest { + #[serde(default)] + display_name: Option, +} + +#[derive(Serialize)] +struct RegisterOptionsResponse { + registration_id: String, + options: CreationChallengeResponse, +} + +#[derive(Deserialize)] +struct RegisterVerifyRequest { + registration_id: String, + credential: RegisterPublicKeyCredential, +} + +#[derive(Serialize)] +struct RegisterVerifyResponse { + tenant_id: String, + credential_id: String, +} + +/// What an interaction challenge is issued for. +/// +/// `deny_unknown_fields` is load-bearing: this route used to take a +/// `body_sha256`, and a client still sending one must be told that it means +/// nothing now rather than have it quietly ignored. An approval that names a +/// body it does not bind would be worse than one that never claimed to. +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +struct RequestOptionsRequest { + /// Which passkey the client intends to use. The runtime allows exactly + /// that one, so an assertion from any other credential fails even before + /// the credential lookup. + credential_id: String, + method: String, + path: String, + #[serde(default)] + query: Option, +} + +/// The assertion, coming back to be turned into a token. +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +struct RequestVerifyRequest { + challenge_id: String, + /// Base64url of the `PublicKeyCredential` JSON `navigator.credentials.get()` + /// produced. + assertion: String, +} + +/// The token, and how long it has to be spent. +#[derive(Serialize)] +struct TokenResponse { + token: String, + /// Seconds. This bounds the time to *start* an interaction, and is not how + /// long one may run once started. + expires_in_secs: u64, +} + +#[derive(Serialize)] +struct RequestOptionsResponse { + challenge_id: String, + options: RequestChallengeResponse, +} + +struct PendingRegistration { + state: PasskeyRegistration, + /// Set when an existing tenant is adding a passkey rather than a new + /// tenant being created. + join: Option<[u8; 16]>, + expires: Instant, +} + +impl std::fmt::Debug for AuthEndpoints { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("AuthEndpoints") + .field( + "pending_registrations", + &self.registrations.lock().map(|r| r.len()).unwrap_or(0), + ) + .finish_non_exhaustive() + } +} + +pub struct AuthEndpoints { + gate: Arc, + credentials: Arc, + entropy: Arc, + fs: Arc, + registrations: Mutex>, +} + +impl AuthEndpoints { + pub fn new( + gate: Arc, + credentials: Arc, + fs: Arc, + entropy: Arc, + ) -> Self { + AuthEndpoints { + gate, + credentials, + entropy, + fs, + registrations: Mutex::new(HashMap::new()), + } + } + + fn random_id(&self) -> anyhow::Result<[u8; 16]> { + let mut id = [0u8; 16]; + self.entropy.get_random(&mut id)?; + Ok(id) + } + + /// Answer an `/auth/*` request, or `None` if this is not one. + pub async fn handle( + &self, + method: &hyper::Method, + path: &str, + body: &[u8], + ) -> Option> { + if !path.starts_with(AUTH_PREFIX) { + return None; + } + if method != hyper::Method::POST { + return Some(problem( + hyper::StatusCode::METHOD_NOT_ALLOWED, + "auth routes are POST", + )); + } + if body.len() > MAX_BODY { + return Some(problem( + hyper::StatusCode::PAYLOAD_TOO_LARGE, + "request body too large", + )); + } + + Some(match &path[AUTH_PREFIX.len()..] { + "register/options" => self.register_options(body).await, + "register/verify" => self.register_verify(body).await, + "request/options" => self.request_options(body).await, + "request/verify" => self.request_verify(body).await, + other => problem( + hyper::StatusCode::NOT_FOUND, + &format!("no auth endpoint {other:?}"), + ), + }) + } + + async fn register_options( + &self, + body: &[u8], + ) -> hyper::Response { + let Ok(request) = serde_json::from_slice::(body) else { + return problem(hyper::StatusCode::BAD_REQUEST, "malformed request"); + }; + + // Registration is open: anyone who can reach this route may create a + // tenant. What that grants is deliberately narrow — a *new*, empty + // tenant and nothing else. It cannot reach an existing tenant's data, + // approve anything, or add a passkey to somebody else's account, all of + // which still need an assertion from a credential already registered + // there. Admission control over resource creation is what was given up + // here; access control was not. + let name = request.display_name.as_deref().unwrap_or("cosigner"); + // A discoverable platform passkey (`residentKey: required`, + // `authenticatorAttachment: platform`), not webauthn-rs's default of + // `residentKey: discouraged` with no attachment. Android below 14 takes + // the default literally and makes a non-discoverable security-key + // credential outside Password Manager, which sign-in then cannot find. + // Clients are native apps, never a browser, so nothing is given up by + // ruling out roaming security keys and cross-device registration. It + // also drops credProtect, which Android does not support and a platform + // passkey does not need. + let (options, state) = match self + .gate + .webauthn() + .start_google_passkey_in_google_password_manager_only_registration( + Uuid::new_v4(), + name, + name, + None, + ) { + Ok(pair) => pair, + Err(e) => { + tracing::error!(error = %e, "starting registration"); + return problem(hyper::StatusCode::INTERNAL_SERVER_ERROR, "unavailable"); + } + }; + + let Ok(id) = self.random_id() else { + return problem(hyper::StatusCode::INTERNAL_SERVER_ERROR, "unavailable"); + }; + { + let now = Instant::now(); + let mut pending = self.registrations.lock().expect("registrations poisoned"); + pending.retain(|_, p| p.expires > now); + if pending.len() >= MAX_REGISTRATIONS { + return problem(hyper::StatusCode::SERVICE_UNAVAILABLE, "try again"); + } + pending.insert( + id, + PendingRegistration { + state, + join: None, + expires: now + REGISTRATION_TTL, + }, + ); + } + + json( + hyper::StatusCode::OK, + &RegisterOptionsResponse { + registration_id: b64(&id), + options, + }, + ) + } + + async fn register_verify( + &self, + body: &[u8], + ) -> hyper::Response { + let Ok(request) = serde_json::from_slice::(body) else { + return problem(hyper::StatusCode::BAD_REQUEST, "malformed request"); + }; + let Some(id) = decode_id(&request.registration_id) else { + return problem(hyper::StatusCode::BAD_REQUEST, "malformed registration id"); + }; + + // Taken, not borrowed: an attempt that fails must not leave the + // registration open for another try at the same challenge. + let pending = { + let mut registrations = self.registrations.lock().expect("registrations poisoned"); + registrations.remove(&id) + }; + let Some(pending) = pending.filter(|p| p.expires > Instant::now()) else { + return problem(hyper::StatusCode::FORBIDDEN, "registration expired"); + }; + + let passkey = match self + .gate + .webauthn() + .finish_passkey_registration(&request.credential, &pending.state) + { + Ok(passkey) => passkey, + Err(e) => { + tracing::warn!(error = %e, "registration did not verify"); + return problem(hyper::StatusCode::FORBIDDEN, "registration did not verify"); + } + }; + + // Joining an existing tenant, or minting one. Minted from the NSM so a + // host cannot predict a tenant id and create its directory first. + let tenant_id = match pending.join { + Some(existing) => existing, + None => match super::credential::mint_tenant_id(&self.entropy) { + Ok(id) => id, + Err(e) => { + tracing::error!(error = format!("{e:#}"), "minting a tenant id"); + return problem(hyper::StatusCode::INTERNAL_SERVER_ERROR, "unavailable"); + } + }, + }; + + let credential_id = request.credential.raw_id.as_ref().to_vec(); + let record = StoredCredential { + version: 1, + tenant_id, + passkey, + active: true, + created_ms: 0, + counter: 0, + }; + if let Err(e) = self.credentials.register(&credential_id, record).await { + tracing::error!(error = format!("{e:#}"), "storing a credential"); + return problem(hyper::StatusCode::CONFLICT, "could not register"); + } + + // The directory, so the tenant's first request finds one rather than + // paying for it while a user waits. + if let Err(e) = crate::tenant::tenant_root_by_id(&self.fs, tenant_id).await { + tracing::error!(error = format!("{e:#}"), "creating a tenant directory"); + return problem(hyper::StatusCode::INTERNAL_SERVER_ERROR, "unavailable"); + } + + tracing::info!( + tenant = %hex::encode(tenant_id), + credential = %hex::encode(&credential_id[..8.min(credential_id.len())]), + joined = pending.join.is_some(), + "registered a passkey" + ); + json( + hyper::StatusCode::OK, + &RegisterVerifyResponse { + tenant_id: hex::encode(tenant_id), + credential_id: b64(&credential_id), + }, + ) + } + + /// Turn a verified assertion into a token for one interaction. + /// + /// The second of the three trips a signed interaction takes, and it has to + /// be its own exchange: a passkey is a challenge-response, so the assertion + /// cannot exist until `request/options` has already answered. Named to + /// match `register/options` → `register/verify`, which is the same shape. + /// + /// **What the token authorizes: one interaction at the method, path and + /// query the challenge was issued for.** It commits to no bytes. A person + /// approved *doing this thing*, not *sending these bytes*, and nothing here + /// should be described as approval of a payload. + async fn request_verify( + &self, + body: &[u8], + ) -> hyper::Response { + let Ok(request) = serde_json::from_slice::(body) else { + return problem(hyper::StatusCode::BAD_REQUEST, "malformed request"); + }; + let Some(challenge_id) = decode_id(&request.challenge_id) else { + return problem(hyper::StatusCode::BAD_REQUEST, "malformed challenge id"); + }; + let Some(assertion_json) = decode(&request.assertion) else { + return problem(hyper::StatusCode::BAD_REQUEST, "malformed assertion"); + }; + let Ok(assertion) = serde_json::from_slice::(&assertion_json) else { + return problem(hyper::StatusCode::BAD_REQUEST, "malformed assertion"); + }; + + let who = match self.gate.authenticate(&challenge_id, &assertion).await { + Ok(who) => who, + Err(denied) => { + // Logged in full, answered in one sentence — the same rule the + // gate has always followed. + tracing::info!(reason = %denied, "refused an assertion"); + return problem(denied.status(), denied.public_message()); + } + }; + + // Minted here, from the enclave's entropy, and held in exactly two + // places: this response, and a hash in the store. + let mut token = [0u8; 32]; + if self.entropy.get_random(&mut token).is_err() { + return problem(hyper::StatusCode::INTERNAL_SERVER_ERROR, "unavailable"); + } + let token = b64(&token); + + let ttl = match self.gate.grant(token.as_bytes(), &who) { + Ok(ttl) => ttl, + Err(denied) => { + tracing::warn!(reason = %denied, "could not record an approval"); + return problem(denied.status(), denied.public_message()); + } + }; + + // The token itself is never logged. What identifies this event is the + // tenant it belongs to and the interaction it is good for. + tracing::info!( + tenant = %hex::encode(who.tenant_id), + method = %who.scope.method, + path = %who.scope.path, + "issued an interaction token" + ); + + json( + hyper::StatusCode::OK, + &TokenResponse { + token, + expires_in_secs: ttl.as_secs(), + }, + ) + } + + async fn request_options( + &self, + body: &[u8], + ) -> hyper::Response { + let Ok(request) = serde_json::from_slice::(body) else { + return problem(hyper::StatusCode::BAD_REQUEST, "malformed request"); + }; + let Some(credential_id) = decode(&request.credential_id) else { + return problem(hyper::StatusCode::BAD_REQUEST, "malformed credential id"); + }; + + let record = match self.credentials_lookup(&credential_id).await { + Ok(Some(record)) if record.active => record, + Ok(_) => { + // Deliberately the same answer as a credential that exists but + // is revoked, and the same as one that never did. + return problem( + hyper::StatusCode::FORBIDDEN, + "no challenge for that credential", + ); + } + Err(e) => { + tracing::error!(error = format!("{e:#}"), "looking up a credential"); + return problem(hyper::StatusCode::INTERNAL_SERVER_ERROR, "unavailable"); + } + }; + + let Ok(id) = self.random_id() else { + return problem(hyper::StatusCode::INTERNAL_SERVER_ERROR, "unavailable"); + }; + let scope = InteractionScope::new(&request.method, &request.path, request.query.as_deref()); + match self + .gate + .issue(id, scope, std::slice::from_ref(&record.passkey)) + { + Ok(options) => json( + hyper::StatusCode::OK, + &RequestOptionsResponse { + challenge_id: b64(&id), + options, + }, + ), + Err(e) => { + tracing::warn!(error = %e, "issuing a challenge"); + problem(hyper::StatusCode::SERVICE_UNAVAILABLE, "try again") + } + } + } + + async fn credentials_lookup( + &self, + credential_id: &[u8], + ) -> anyhow::Result> { + use super::gate::CredentialStore; + self.credentials.lookup(credential_id).await + } +} + +fn b64(bytes: &[u8]) -> String { + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(bytes) +} + +fn decode(text: &str) -> Option> { + base64::engine::general_purpose::URL_SAFE_NO_PAD + .decode(text.trim()) + .ok() +} + +fn decode_id(text: &str) -> Option<[u8; 16]> { + decode(text).and_then(|b| <[u8; 16]>::try_from(b.as_slice()).ok()) +} + +fn body_of(bytes: Vec) -> wasmtime_wasi_http::p2::body::HyperOutgoingBody { + use http_body_util::BodyExt; + http_body_util::Full::new(Bytes::from(bytes)) + .map_err(|e: std::convert::Infallible| match e {}) + .boxed_unsync() +} + +fn json( + status: hyper::StatusCode, + value: &T, +) -> hyper::Response { + let bytes = serde_json::to_vec(value).unwrap_or_else(|_| b"{}".to_vec()); + hyper::Response::builder() + .status(status) + .header("content-type", "application/json") + // Challenges and options are single-use and short-lived; a cache that + // replayed one would be handing out a used challenge. + .header("cache-control", "no-store") + .body(body_of(bytes)) + .expect("response is well formed") +} + +fn problem( + status: hyper::StatusCode, + detail: &str, +) -> hyper::Response { + json(status, &serde_json::json!({ "error": detail })) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::auth::challenge::{ChallengeStore, DEFAULT_CAPACITY, DEFAULT_TTL}; + use crate::auth::testing::{Relying, ORIGIN, RP_ID}; + use crate::auth::SoftwareAuthenticator; + use http_body_util::BodyExt; + use s3fs_core::backend::memory::MemoryBackend; + use s3fs_core::{Config, Fs, MasterSecret}; + + struct Fixture { + endpoints: AuthEndpoints, + fs: Arc, + credentials: Arc, + } + + async fn fixture() -> Fixture { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([7u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("filesystem"); + + let credentials = Arc::new(FilesystemCredentials::new(fs.clone())); + let gate = Arc::new(Gate::new( + Relying::new().webauthn, + ChallengeStore::new(DEFAULT_TTL, DEFAULT_CAPACITY), + credentials.clone(), + super::super::TokenStore::new( + std::time::Duration::from_secs(60), + super::super::token::DEFAULT_CAPACITY, + ), + )); + let entropy: Arc = Arc::new(nitro_nsm::fake::FakeNsm::new()); + let endpoints = AuthEndpoints::new(gate, credentials.clone(), fs.clone(), entropy); + Fixture { + endpoints, + fs, + credentials, + } + } + + async fn post(f: &Fixture, path: &str, body: serde_json::Value) -> (u16, serde_json::Value) { + let response = f + .endpoints + .handle(&hyper::Method::POST, path, body.to_string().as_bytes()) + .await + .expect("an auth route"); + let status = response.status().as_u16(); + let bytes = response.into_body().collect().await.unwrap().to_bytes(); + ( + status, + serde_json::from_slice(&bytes).unwrap_or(serde_json::Value::Null), + ) + } + + /// Register a passkey the way a client does, and get its tenant back. + async fn register(f: &Fixture, auth: &SoftwareAuthenticator) -> (u16, serde_json::Value) { + let (status, options) = post(f, "/auth/register/options", serde_json::json!({})).await; + if status != 200 { + return (status, options); + } + let challenge = options["options"]["publicKey"]["challenge"] + .as_str() + .expect("a challenge") + .to_string(); + post( + f, + "/auth/register/verify", + serde_json::json!({ + "registration_id": options["registration_id"], + "credential": auth.register(&challenge, ORIGIN), + }), + ) + .await + } + + /// Options ask for a passkey sign-in can find. With webauthn-rs's defaults, + /// Android below 14 made a credential One Tap reports as "Cannot find a + /// matching credential". + #[tokio::test] + async fn registration_asks_for_a_discoverable_platform_passkey() { + let f = fixture().await; + let (status, body) = post(&f, "/auth/register/options", serde_json::json!({})).await; + assert_eq!(status, 200, "{body}"); + let selection = &body["options"]["publicKey"]["authenticatorSelection"]; + assert_eq!(selection["residentKey"], "required", "{selection}"); + assert_eq!(selection["requireResidentKey"], true, "{selection}"); + assert_eq!( + selection["authenticatorAttachment"], "platform", + "{selection}" + ); + assert_eq!(selection["userVerification"], "required", "{selection}"); + } + + /// The whole registration path: options, response, tenant. + #[tokio::test] + async fn registration_creates_a_tenant() { + let f = fixture().await; + let auth = SoftwareAuthenticator::new(RP_ID); + let (status, body) = register(&f, &auth).await; + assert_eq!(status, 200, "{body}"); + + let tenant_hex = body["tenant_id"].as_str().expect("a tenant id"); + assert_eq!(tenant_hex.len(), 32, "a 16-byte tenant id in hex"); + + // The credential is stored and points at that tenant. + use crate::auth::gate::CredentialStore; + let record = f + .credentials + .lookup(auth.credential_id()) + .await + .unwrap() + .expect("the credential was stored"); + assert_eq!(hex::encode(record.tenant_id), tenant_hex); + + // And the tenant's directory exists, so their first request is cheap. + assert!(crate::tenant::tenants(&f.fs) + .await + .unwrap() + .contains(&tenant_hex.to_string())); + } + + /// Registration is open, and each one stands alone: a second, unrelated + /// passkey registers too and lands in a tenant of its own rather than + /// joining the first. + #[tokio::test] + async fn registration_is_open_and_each_one_is_a_new_tenant() { + let f = fixture().await; + let (first_status, first) = register(&f, &SoftwareAuthenticator::new(RP_ID)).await; + assert_eq!(first_status, 200, "{first}"); + let (second_status, second) = register(&f, &SoftwareAuthenticator::new(RP_ID)).await; + assert_eq!(second_status, 200, "{second}"); + assert_ne!( + first["tenant_id"], second["tenant_id"], + "a second registration joined the first one's tenant" + ); + } + + /// An older client still sending an enrollment token is told the field + /// means nothing now, rather than being quietly admitted as though one had + /// been checked. + #[tokio::test] + async fn a_registration_still_carrying_an_enrollment_token_is_refused() { + let f = fixture().await; + let (status, _) = post( + &f, + "/auth/register/options", + serde_json::json!({ "enrollment_token": "not-a-real-token-but-long" }), + ) + .await; + assert_eq!(status, 400); + } + + /// A registration response for a challenge that was never issued. + #[tokio::test] + async fn an_unknown_registration_id_is_refused() { + let f = fixture().await; + let auth = SoftwareAuthenticator::new(RP_ID); + let (status, _) = post( + &f, + "/auth/register/verify", + serde_json::json!({ + "registration_id": "AAAAAAAAAAAAAAAAAAAAAA", + "credential": auth.register("Y2hhbGxlbmdl", ORIGIN), + }), + ) + .await; + assert_eq!(status, 403); + } + + /// A failed attempt must not leave the registration open for another try. + #[tokio::test] + async fn a_registration_id_cannot_be_retried() { + let f = fixture().await; + let (_, options) = post(&f, "/auth/register/options", serde_json::json!({})).await; + let auth = SoftwareAuthenticator::new(RP_ID); + + // Wrong challenge: fails. + let (first, _) = post( + &f, + "/auth/register/verify", + serde_json::json!({ + "registration_id": options["registration_id"], + "credential": auth.register("d3JvbmctY2hhbGxlbmdl", ORIGIN), + }), + ) + .await; + assert_eq!(first, 403); + + // The right one now also fails: the id was consumed by the attempt. + let challenge = options["options"]["publicKey"]["challenge"] + .as_str() + .unwrap(); + let (second, _) = post( + &f, + "/auth/register/verify", + serde_json::json!({ + "registration_id": options["registration_id"], + "credential": auth.register(challenge, ORIGIN), + }), + ) + .await; + assert_eq!(second, 403, "a failed registration could be retried"); + } + + /// A challenge is issued only for a credential this runtime knows. + #[tokio::test] + async fn a_challenge_is_issued_for_a_registered_credential() { + let f = fixture().await; + let auth = SoftwareAuthenticator::new(RP_ID); + assert_eq!(register(&f, &auth).await.0, 200); + + let b64 = base64::engine::general_purpose::URL_SAFE_NO_PAD; + let (status, body) = post( + &f, + "/auth/request/options", + serde_json::json!({ + "credential_id": b64.encode(auth.credential_id()), + "method": "POST", + "path": "/sign", + }), + ) + .await; + assert_eq!(status, 200, "{body}"); + assert!(body["challenge_id"].is_string()); + assert!(body["options"]["publicKey"]["challenge"].is_string()); + } + + /// Unknown and revoked credentials are refused identically, so the route + /// cannot be used to learn which passkeys exist. + #[tokio::test] + async fn unknown_and_revoked_credentials_are_indistinguishable() { + let f = fixture().await; + let auth = SoftwareAuthenticator::new(RP_ID); + assert_eq!(register(&f, &auth).await.0, 200); + f.credentials.revoke(auth.credential_id()).await.unwrap(); + + async fn ask(f: &Fixture, id: &[u8]) -> (u16, serde_json::Value) { + let b64 = base64::engine::general_purpose::URL_SAFE_NO_PAD; + post( + f, + "/auth/request/options", + serde_json::json!({ + "credential_id": b64.encode(id), + "method": "POST", + "path": "/sign", + }), + ) + .await + } + let revoked = ask(&f, auth.credential_id()).await; + let unknown = ask(&f, &[0xff; 32]).await; + assert_eq!(revoked.0, 403); + assert_eq!(revoked, unknown, "the refusals differ and leak existence"); + } + + /// Everything outside `/auth/` is somebody else's business. + #[tokio::test] + async fn other_paths_are_not_claimed() { + let f = fixture().await; + assert!(f + .endpoints + .handle(&hyper::Method::POST, "/sign", b"{}") + .await + .is_none()); + assert!(f + .endpoints + .handle(&hyper::Method::GET, "/counter", b"") + .await + .is_none()); + } + + #[tokio::test] + async fn an_unknown_auth_route_is_a_404() { + let f = fixture().await; + let (status, _) = post(&f, "/auth/nonsense", serde_json::json!({})).await; + assert_eq!(status, 404); + } +} diff --git a/runtime/src/auth/testing.rs b/runtime/src/auth/testing.rs new file mode 100644 index 0000000..08a5d06 --- /dev/null +++ b/runtime/src/auth/testing.rs @@ -0,0 +1,460 @@ +//! A relying party in a test, paired with [`super::SoftwareAuthenticator`]. +//! +//! The point of this file is one assertion: **bytes the software +//! authenticator produces are bytes `webauthn-rs` accepts.** If that ever +//! stopped being true, every test built on the authenticator would be checking +//! the runtime against a fiction. So the round trip is exercised directly, and +//! the rest of the suite builds on it. +//! +//! The challenge is pulled out of the options the way a browser gets it — +//! serialise to JSON, read `publicKey.challenge` — rather than by reaching +//! into the crate's types. That is the path a real client takes, so it is the +//! path worth depending on. + +use webauthn_rs::prelude::*; + +use super::SoftwareAuthenticator; + +pub const RP_ID: &str = "enclave.test"; +pub const ORIGIN: &str = "https://enclave.test"; + +pub struct Relying { + pub webauthn: Webauthn, +} + +impl Default for Relying { + fn default() -> Self { + Self::new() + } +} + +impl Relying { + pub fn new() -> Self { + Relying { + webauthn: WebauthnBuilder::new(RP_ID, &Url::parse(ORIGIN).expect("origin")) + .expect("relying party") + .build() + .expect("relying party"), + } + } + + /// The challenge a browser would read out of the options it was handed. + fn challenge_of(options: &T) -> String { + serde_json::to_value(options) + .expect("options serialise") + .get("publicKey") + .and_then(|k| k.get("challenge")) + .and_then(|c| c.as_str()) + .expect("options carry a challenge") + .to_string() + } + + /// Register a fresh software passkey and return it, ready to authenticate. + pub fn register(&self, auth: &SoftwareAuthenticator) -> Passkey { + let (options, state) = self + .webauthn + .start_passkey_registration(Uuid::new_v4(), "tester", "Tester", None) + .expect("registration options"); + let response: RegisterPublicKeyCredential = + serde_json::from_value(auth.register(&Self::challenge_of(&options), ORIGIN)) + .expect("the authenticator produces a well-formed registration"); + self.webauthn + .finish_passkey_registration(&response, &state) + .expect("the authenticator's registration verifies") + } + + /// Begin an authentication against a throwaway credential. + /// + /// For tests that only need a `PasskeyAuthentication` to put in the + /// challenge store and do not care which credential it names. + pub fn begin_authentication(&self) -> (RequestChallengeResponse, PasskeyAuthentication) { + let auth = SoftwareAuthenticator::new(RP_ID); + let passkey = self.register(&auth); + self.webauthn + .start_passkey_authentication(std::slice::from_ref(&passkey)) + .expect("authentication options") + } + + /// Challenge, then assert against it, then verify — the whole ceremony. + pub fn round_trip( + &self, + auth: &SoftwareAuthenticator, + passkey: &Passkey, + ) -> WebauthnResult { + let (options, state) = self + .webauthn + .start_passkey_authentication(std::slice::from_ref(passkey)) + .expect("authentication options"); + let response: PublicKeyCredential = + serde_json::from_value(auth.assert(&Self::challenge_of(&options), ORIGIN)) + .expect("the authenticator produces a well-formed assertion"); + self.webauthn + .finish_passkey_authentication(&response, &state) + } + + pub fn challenge_for(options: &RequestChallengeResponse) -> String { + Self::challenge_of(options) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::auth::authenticator::flags; + + /// The foundation. Everything else in the suite trusts that this holds. + #[test] + fn a_software_passkey_registers_and_authenticates() { + let rp = Relying::new(); + let auth = SoftwareAuthenticator::new(RP_ID); + let passkey = rp.register(&auth); + + let result = rp.round_trip(&auth, &passkey).expect("assertion verifies"); + assert_eq!(result.cred_id().as_ref(), auth.credential_id()); + assert!(result.user_verified(), "the fixture must assert UV"); + } + + /// A passkey that merely sat in an unlocked pocket is not approval. The + /// crate is configured `UserVerificationPolicy::Required`; this proves it. + #[test] + fn an_assertion_without_user_verification_is_refused() { + let rp = Relying::new(); + let auth = SoftwareAuthenticator::new(RP_ID); + let passkey = rp.register(&auth); + + let (options, state) = rp + .webauthn + .start_passkey_authentication(std::slice::from_ref(&passkey)) + .unwrap(); + let response: PublicKeyCredential = serde_json::from_value(auth.assert_with( + &Relying::challenge_for(&options), + ORIGIN, + flags::UP, // present, but not verified + )) + .unwrap(); + assert!( + rp.webauthn + .finish_passkey_authentication(&response, &state) + .is_err(), + "a merely-present authenticator must not authorize anything" + ); + } + + /// A page on another origin cannot borrow the user's passkey for this one. + #[test] + fn an_assertion_from_another_origin_is_refused() { + let rp = Relying::new(); + let auth = SoftwareAuthenticator::new(RP_ID); + let passkey = rp.register(&auth); + + let (options, state) = rp + .webauthn + .start_passkey_authentication(std::slice::from_ref(&passkey)) + .unwrap(); + let response: PublicKeyCredential = serde_json::from_value(auth.assert( + &Relying::challenge_for(&options), + "https://attacker.example", + )) + .unwrap(); + assert!(rp + .webauthn + .finish_passkey_authentication(&response, &state) + .is_err()); + } + + /// An Android app claims `android:apk-key-hash:…`, never `https://`. + /// Allowed, it registers and authenticates; any other app is still refused. + #[test] + fn an_allowed_android_app_registers_and_authenticates_and_no_other_app_does() { + const APP: &str = "android:apk-key-hash:47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU"; + const OTHER_APP: &str = "android:apk-key-hash:bjQLnP-zepicpUTmu3gKLHiQHT-zNzh2hRGjBhevoB0"; + let rp = Relying { + webauthn: crate::build_relying_party(RP_ID, ORIGIN, &[APP.into()]) + .expect("relying party"), + }; + let auth = SoftwareAuthenticator::new(RP_ID); + + let (options, state) = rp + .webauthn + .start_passkey_registration(Uuid::new_v4(), "tester", "Tester", None) + .unwrap(); + let response: RegisterPublicKeyCredential = + serde_json::from_value(auth.register(&Relying::challenge_of(&options), APP)).unwrap(); + let passkey = rp + .webauthn + .finish_passkey_registration(&response, &state) + .expect("the app's registration verifies"); + + let assert_from = |origin: &str| { + let (options, state) = rp + .webauthn + .start_passkey_authentication(std::slice::from_ref(&passkey)) + .unwrap(); + let response: PublicKeyCredential = + serde_json::from_value(auth.assert(&Relying::challenge_for(&options), origin)) + .unwrap(); + rp.webauthn.finish_passkey_authentication(&response, &state) + }; + assert_from(APP).expect("the app's assertion verifies"); + assert_from(ORIGIN).expect("the web origin still verifies"); + assert!( + assert_from(OTHER_APP).is_err(), + "an app signed with another key must not borrow the passkey" + ); + } + + /// Without the setting, what the app sends is refused — the blocker this + /// setting exists to lift. + #[test] + fn an_android_app_is_refused_unless_allowed() { + let rp = Relying::new(); + let auth = SoftwareAuthenticator::new(RP_ID); + let passkey = rp.register(&auth); + + let (options, state) = rp + .webauthn + .start_passkey_authentication(std::slice::from_ref(&passkey)) + .unwrap(); + let response: PublicKeyCredential = serde_json::from_value(auth.assert( + &Relying::challenge_for(&options), + "android:apk-key-hash:47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU", + )) + .unwrap(); + assert!(rp + .webauthn + .finish_passkey_authentication(&response, &state) + .is_err()); + } + + /// Signing something other than the challenge it was given. + #[test] + fn an_assertion_over_the_wrong_challenge_is_refused() { + let rp = Relying::new(); + let auth = SoftwareAuthenticator::new(RP_ID); + let passkey = rp.register(&auth); + + let (_, state) = rp + .webauthn + .start_passkey_authentication(std::slice::from_ref(&passkey)) + .unwrap(); + let response: PublicKeyCredential = + serde_json::from_value(auth.assert("bm90LXRoZS1jaGFsbGVuZ2U", ORIGIN)).unwrap(); + assert!(rp + .webauthn + .finish_passkey_authentication(&response, &state) + .is_err()); + } + + /// A different passkey, however well-formed, is not this one. + #[test] + fn another_credential_cannot_answer_this_challenge() { + let rp = Relying::new(); + let mine = SoftwareAuthenticator::new(RP_ID); + let theirs = SoftwareAuthenticator::new(RP_ID); + let passkey = rp.register(&mine); + rp.register(&theirs); + + let (options, state) = rp + .webauthn + .start_passkey_authentication(std::slice::from_ref(&passkey)) + .unwrap(); + let response: PublicKeyCredential = + serde_json::from_value(theirs.assert(&Relying::challenge_for(&options), ORIGIN)) + .unwrap(); + assert!(rp + .webauthn + .finish_passkey_authentication(&response, &state) + .is_err()); + } +} + +/// Credentials in memory, for tests and as the smallest thing that satisfies +/// [`CredentialStore`]. +#[derive(Default)] +pub struct MemoryCredentials { + records: std::sync::Mutex, super::CredentialRecord>>, +} + +impl MemoryCredentials { + pub fn insert(&self, credential_id: &[u8], record: super::CredentialRecord) { + self.records + .lock() + .expect("credentials poisoned") + .insert(credential_id.to_vec(), record); + } + + pub fn revoke(&self, credential_id: &[u8]) { + if let Some(r) = self + .records + .lock() + .expect("credentials poisoned") + .get_mut(credential_id) + { + r.active = false; + } + } + + pub fn counter(&self, credential_id: &[u8]) -> Option { + self.records + .lock() + .expect("credentials poisoned") + .get(credential_id) + .map(|r| r.counter) + } +} + +#[async_trait::async_trait] +impl super::CredentialStore for MemoryCredentials { + async fn lookup( + &self, + credential_id: &[u8], + ) -> anyhow::Result> { + Ok(self + .records + .lock() + .expect("credentials poisoned") + .get(credential_id) + .cloned()) + } + + async fn record_use(&self, credential_id: &[u8], counter: u32) -> anyhow::Result<()> { + if let Some(r) = self + .records + .lock() + .expect("credentials poisoned") + .get_mut(credential_id) + { + r.counter = counter; + } + Ok(()) + } +} + +/// A gate with one registered passkey, ready to be asked awkward questions. +pub struct Harness { + pub gate: super::Gate, + pub authenticator: SoftwareAuthenticator, + pub passkey: Passkey, + pub credentials: std::sync::Arc, + pub tenant_id: [u8; 16], +} + +impl Default for Harness { + fn default() -> Self { + Self::new() + } +} + +impl Harness { + pub fn new() -> Self { + Self::with_token_ttl(super::token::DEFAULT_TTL) + } + + pub fn with_token_ttl(ttl: std::time::Duration) -> Self { + let rp = Relying::new(); + let authenticator = SoftwareAuthenticator::new(RP_ID); + let passkey = rp.register(&authenticator); + let tenant_id = [0xab; 16]; + + let credentials = std::sync::Arc::new(MemoryCredentials::default()); + credentials.insert( + authenticator.credential_id(), + super::CredentialRecord { + tenant_id, + passkey: passkey.clone(), + active: true, + counter: 0, + }, + ); + + Harness { + gate: super::Gate::new( + rp.webauthn, + super::ChallengeStore::new(super::DEFAULT_TTL, super::DEFAULT_CAPACITY), + credentials.clone(), + super::TokenStore::new(ttl, super::token::DEFAULT_CAPACITY), + ), + authenticator, + passkey, + credentials, + tenant_id, + } + } + + /// Walk the real two-trip flow and return the token it produces. + /// + /// Challenge, assertion, verification — the same path a client takes, so a + /// test that uses this is exercising what a client actually does rather + /// than a shortcut around it. + pub async fn token_for(&self, method: &str, path_and_query: &str) -> String { + let (path, query) = match path_and_query.split_once('?') { + Some((p, q)) => (p, Some(q)), + None => (path_and_query, None), + }; + let id = [7u8; 16]; + let options = self + .gate + .issue( + id, + super::InteractionScope::new(method, path, query), + std::slice::from_ref(&self.passkey), + ) + .expect("issuing a challenge"); + let assertion = self + .authenticator + .assert(&Relying::challenge_for(&options), ORIGIN); + let assertion: webauthn_rs::prelude::PublicKeyCredential = + serde_json::from_str(&assertion.to_string()).expect("a credential"); + let who = self + .gate + .authenticate(&id, &assertion) + .await + .expect("the assertion verifies"); + + let token = "a-test-token-of-plausible-length".to_string(); + self.gate + .grant(token.as_bytes(), &who) + .expect("recording the approval"); + token + } + + /// A request carrying a bearer token. + pub fn bearer( + method: &str, + path_and_query: &str, + body: &[u8], + token: &str, + ) -> hyper::Request> { + hyper::Request::builder() + .method(method) + .uri(format!("https://enclave.test{path_and_query}")) + .header(super::AUTHORIZATION_HEADER, format!("Bearer {token}")) + .body(http_body_util::Full::new(bytes::Bytes::copy_from_slice( + body, + ))) + .expect("well-formed request") + } + + pub fn request_with( + method: &str, + path_and_query: &str, + body: &[u8], + challenge_id: &[u8; 16], + assertion: &serde_json::Value, + ) -> hyper::Request> { + use base64::Engine as _; + let b64 = base64::engine::general_purpose::URL_SAFE_NO_PAD; + hyper::Request::builder() + .method(method) + .uri(format!("https://{RP_ID}{path_and_query}")) + .header(super::gate::CHALLENGE_HEADER, b64.encode(challenge_id)) + .header( + super::gate::ASSERTION_HEADER, + b64.encode(assertion.to_string()), + ) + .body(http_body_util::Full::new(bytes::Bytes::copy_from_slice( + body, + ))) + .expect("well-formed request") + } +} diff --git a/runtime/src/auth/token.rs b/runtime/src/auth/token.rs new file mode 100644 index 0000000..fa22917 --- /dev/null +++ b/runtime/src/auth/token.rs @@ -0,0 +1,319 @@ +//! Interaction tokens: what a passkey approval buys, once. +//! +//! A person approves an **interaction** — one HTTP request and its response, or +//! one bidirectional stream until it closes or reaches its lifetime limit. The +//! runtime records that approval as a short-lived, single-use bearer token, and +//! the interaction presents it. +//! +//! # What a token is and is not +//! +//! It authorizes **one interaction at one route**: a method, a path and a +//! query. It says nothing whatever about the bytes that travel on it. +//! +//! That is a deliberate policy, and the honest way to describe it is that the +//! approval names *what the person is about to do*, not *what they are about to +//! send*. A token issued for `POST /sign` authorizes whatever body follows, so +//! a client compromised between the approval and the request can substitute the +//! payload. Nothing here should be described as proof that a person approved +//! particular bytes, because it is not. +//! +//! What it does still carry: a person authenticated with their passkey to get +//! it, it belongs to exactly one tenant, it is good once, and it expires. +//! +//! # Why the store never holds a token +//! +//! Entries are keyed by `sha256(token)`. Possession of the store is not +//! possession of a token — the same reason a password file stores hashes +//! rather than the passwords themselves. +//! +//! Nothing here survives a restart, and nothing should: an approval that +//! outlived the enclave that issued it would be an approval nobody could +//! account for. + +use std::collections::HashMap; +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +/// How long a token may sit unused before it is worthless. +/// +/// This bounds the time to *start* an interaction, and is deliberately not the +/// same thing as how long one may run once started — see the runtime's +/// interaction deadline for that. A person who approves something and then puts +/// their phone down should not find the approval still live an hour later. +pub const DEFAULT_TTL: Duration = Duration::from_secs(60); + +/// Outstanding tokens allowed at once. +/// +/// The bound on what authenticated callers can make the runtime hold. Reaching +/// it refuses new tokens rather than evicting live ones — evicting to admit +/// would let one caller cancel another's approval. +pub const DEFAULT_CAPACITY: usize = 1024; + +/// The one interaction a token is good for. +/// +/// Every field is something an attacker would otherwise be free to vary while +/// spending an approval given for something else. There is deliberately no +/// body hash: an interaction may be a stream, whose body does not exist when +/// the approval is given. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct InteractionScope { + pub method: String, + pub path: String, + /// `None` and `Some("")` are different: an approval for `?a=1` must not be + /// usable for a request with no query at all. + pub query: Option, +} + +impl InteractionScope { + pub fn new(method: &str, path: &str, query: Option<&str>) -> Self { + InteractionScope { + method: method.to_ascii_uppercase(), + path: path.to_string(), + query: query.map(str::to_string), + } + } +} + +/// What an approval was recorded as. +struct Granted { + tenant_id: [u8; 16], + scope: InteractionScope, + expires: Instant, +} + +/// Why a token could not be spent. +/// +/// Distinguished for the log, never for the client. "Wrong route" and "no such +/// token" tell an attacker which half of a guess was right; a legitimate caller +/// learns nothing it can act on from either. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum TokenError { + Unknown, + Expired, + WrongInteraction, + Full, +} + +impl std::fmt::Display for TokenError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + TokenError::Unknown => "no such token, or it has already been spent", + TokenError::Expired => "the token expired before it was used", + TokenError::WrongInteraction => "the token was issued for a different interaction", + TokenError::Full => "too many outstanding tokens", + }) + } +} + +/// Live approvals, keyed by the hash of the token that names them. +/// +/// No `Debug`: a store that printed itself would put the only thing standing +/// between a caller and a tenant's guest into any log line that formatted a +/// struct containing one. +pub struct TokenStore { + granted: Mutex>, + ttl: Duration, + capacity: usize, +} + +impl TokenStore { + pub fn new(ttl: Duration, capacity: usize) -> Self { + TokenStore { + granted: Mutex::new(HashMap::new()), + ttl, + capacity: capacity.max(1), + } + } + + pub fn ttl(&self) -> Duration { + self.ttl + } + + /// Record an approval against a token the caller has already generated. + /// + /// The token itself is never stored — only its hash — so this takes the + /// bytes, hashes them, and forgets them. + pub fn issue( + &self, + token: &[u8], + tenant_id: [u8; 16], + scope: InteractionScope, + ) -> Result<(), TokenError> { + let now = Instant::now(); + let mut granted = self.granted.lock().expect("token store poisoned"); + granted.retain(|_, g| g.expires > now); + if granted.len() >= self.capacity { + return Err(TokenError::Full); + } + granted.insert( + nitro_attestation::sha256(token), + Granted { + tenant_id, + scope, + expires: now + self.ttl, + }, + ); + Ok(()) + } + + /// Spend a token on one interaction. + /// + /// Removed before anything about it is checked, so a token offered for the + /// wrong route is gone as surely as one that worked. Two callers racing the + /// same token reach the `remove` in some order and exactly one finds it — + /// which is what makes "one interaction" true under concurrency rather than + /// only in a diagram. + pub fn redeem(&self, token: &[u8], scope: &InteractionScope) -> Result<[u8; 16], TokenError> { + let mut granted = self.granted.lock().expect("token store poisoned"); + let entry = granted + .remove(&nitro_attestation::sha256(token)) + .ok_or(TokenError::Unknown)?; + if entry.expires <= Instant::now() { + return Err(TokenError::Expired); + } + if &entry.scope != scope { + return Err(TokenError::WrongInteraction); + } + Ok(entry.tenant_id) + } + + pub fn outstanding(&self) -> usize { + let now = Instant::now(); + let mut granted = self.granted.lock().expect("token store poisoned"); + granted.retain(|_, g| g.expires > now); + granted.len() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn scope() -> InteractionScope { + InteractionScope::new("POST", "/sign", None) + } + + fn store() -> TokenStore { + TokenStore::new(DEFAULT_TTL, DEFAULT_CAPACITY) + } + + #[test] + fn a_token_is_good_for_the_interaction_it_was_issued_for() { + let s = store(); + s.issue(b"a-token", [7u8; 16], scope()).unwrap(); + assert_eq!(s.redeem(b"a-token", &scope()).unwrap(), [7u8; 16]); + } + + /// **One interaction.** The second attempt has nothing to find. + #[test] + fn a_token_cannot_be_spent_twice() { + let s = store(); + s.issue(b"a-token", [7u8; 16], scope()).unwrap(); + s.redeem(b"a-token", &scope()).unwrap(); + assert_eq!(s.redeem(b"a-token", &scope()), Err(TokenError::Unknown)); + } + + /// **The substitution this model still refuses.** An approval names a + /// route, and cannot be moved to another one. + #[test] + fn a_token_cannot_be_moved_to_another_route() { + for wrong in [ + InteractionScope::new("POST", "/withdraw", None), + InteractionScope::new("GET", "/sign", None), + InteractionScope::new("POST", "/sign", Some("all=true")), + ] { + let s = store(); + s.issue(b"a-token", [7u8; 16], scope()).unwrap(); + assert_eq!( + s.redeem(b"a-token", &wrong), + Err(TokenError::WrongInteraction), + "an approval for {:?} was spent on {wrong:?}", + scope() + ); + } + } + + /// And a refused attempt still spends it — a caller that guessed the route + /// wrong does not get to keep guessing with the same token. + #[test] + fn a_token_offered_for_the_wrong_route_is_still_gone() { + let s = store(); + s.issue(b"a-token", [7u8; 16], scope()).unwrap(); + let wrong = InteractionScope::new("POST", "/withdraw", None); + assert_eq!( + s.redeem(b"a-token", &wrong), + Err(TokenError::WrongInteraction) + ); + assert_eq!( + s.redeem(b"a-token", &scope()), + Err(TokenError::Unknown), + "the token survived being offered for the wrong interaction" + ); + } + + #[test] + fn an_expired_token_is_refused_and_destroyed_by_the_attempt() { + let s = TokenStore::new(Duration::from_millis(1), DEFAULT_CAPACITY); + s.issue(b"a-token", [7u8; 16], scope()).unwrap(); + std::thread::sleep(Duration::from_millis(5)); + assert_eq!(s.redeem(b"a-token", &scope()), Err(TokenError::Expired)); + assert_eq!(s.redeem(b"a-token", &scope()), Err(TokenError::Unknown)); + } + + /// A full store refuses new approvals rather than cancelling live ones. + #[test] + fn a_full_store_refuses_rather_than_evicting() { + let s = TokenStore::new(DEFAULT_TTL, 2); + s.issue(b"one", [1u8; 16], scope()).unwrap(); + s.issue(b"two", [2u8; 16], scope()).unwrap(); + assert_eq!(s.issue(b"three", [3u8; 16], scope()), Err(TokenError::Full)); + // And the two that were there are untouched. + assert_eq!(s.redeem(b"one", &scope()).unwrap(), [1u8; 16]); + assert_eq!(s.redeem(b"two", &scope()).unwrap(), [2u8; 16]); + } + + /// Two tenants' approvals never resolve to each other. + #[test] + fn a_token_resolves_only_to_its_own_tenant() { + let s = store(); + s.issue(b"alices", [0xaa; 16], scope()).unwrap(); + s.issue(b"bobs", [0xbb; 16], scope()).unwrap(); + assert_eq!(s.redeem(b"alices", &scope()).unwrap(), [0xaa; 16]); + assert_eq!(s.redeem(b"bobs", &scope()).unwrap(), [0xbb; 16]); + } + + /// Nothing outlives the enclave that issued it. + #[test] + fn an_unused_token_does_not_survive_a_restart() { + let before = store(); + before.issue(b"a-token", [7u8; 16], scope()).unwrap(); + let after = store(); + assert_eq!( + after.redeem(b"a-token", &scope()), + Err(TokenError::Unknown), + "an approval survived the runtime that issued it" + ); + } + + /// **Under concurrency, exactly one winner.** + #[test] + fn racing_redemptions_have_exactly_one_winner() { + use std::sync::Arc; + for _ in 0..25 { + let s = Arc::new(store()); + s.issue(b"a-token", [7u8; 16], scope()).unwrap(); + let winners: usize = std::thread::scope(|scope_| { + let handles: Vec<_> = (0..8) + .map(|_| { + let s = s.clone(); + scope_.spawn(move || { + usize::from(s.redeem(b"a-token", &super::tests::scope()).is_ok()) + }) + }) + .collect(); + handles.into_iter().map(|h| h.join().unwrap()).sum() + }); + assert_eq!(winners, 1, "a token was spent {winners} times"); + } + } +} diff --git a/runtime/src/bin/passkey-client.rs b/runtime/src/bin/passkey-client.rs new file mode 100644 index 0000000..5138272 --- /dev/null +++ b/runtime/src/bin/passkey-client.rs @@ -0,0 +1,1265 @@ +//! `passkey-client` — drive the enclave's WebAuthn gate from a script. +//! +//! ```console +//! $ passkey-client --url https://127.0.0.1:8443 enrol +//! $ passkey-client --url https://127.0.0.1:8443 get --path /counter +//! ``` +//! +//! A phone cannot be driven from a shell script, and the QEMU harness serves a +//! self-signed certificate that no platform authenticator would attest against +//! anyway. So this is the client half in software: it registers a P-256 passkey +//! and then, for each request, asks for a challenge bound to that exact request +//! and signs it — the same two round trips a real app makes, producing the same +//! bytes. +//! +//! It is a *client*, not a bypass. Everything it sends goes through the same +//! gate as anything else, and the enclave cannot tell the difference — which is +//! the point: if the harness can reach the guest with this, the gate is wired +//! up, and if it cannot without it, the gate is doing its job. +//! +//! # The order of a signed request +//! +//! ```text +//! 1. POST /auth/request/options ── carries the body hash, no approval +//! 2. check that response's document against this connection ◀── refuse here +//! and remember the certificate it was served under +//! 3. ask the authenticator to sign (a person is prompted) +//! 4. open the operation's connection, complete the handshake, and check it +//! presents that same certificate ◀── refuse *before* sending +//! 5. only then send the operation +//! ``` +//! +//! Step 2 is why there is no separate probe route. Verifying a connection means +//! receiving a response on it, so *something* has to go first on a connection +//! nothing has vouched for — and the challenge request is the right thing to +//! send: it carries the operation's hash but no approval, so a party that +//! intercepted it learns what is intended and holds nothing it can act on. The +//! round trip was needed anyway. +//! +//! Step 2 comes before step 3 deliberately. A person should not be asked to +//! approve an operation for a party nobody has identified yet. +//! +//! Step 4 needs no second attestation document, and that is the point. The +//! document from step 2 binds a certificate; TLS proves the peer holds that +//! certificate's *private key*, which the certificate itself — being public — +//! does not. So "the same certificate" and "the same enclave" are the same +//! statement, and checking it costs a comparison rather than a signature. +//! +//! It is checked after the handshake and before a byte is written, so a changed +//! far end means the operation is never sent. Nothing is retried on a fresh +//! connection: the assertion is bound to this exact body and is single-use, so +//! a retry would risk executing twice for the sake of reaching a party that has +//! just failed to be the one approved. +//! +//! This rests on the runtime refusing session resumption — see `serve::tls`. A +//! resumed session need present no certificate at all, and pinning against one +//! the client had cached would be checking its own memory. +//! +//! What none of this establishes: the document is generated before the request +//! is routed, so it says nothing about the response body and does not show the +//! guest ran. Signature and chain are `nitro-attest`'s to check, and need a +//! trust root this client is not given. +//! +//! The credential is kept in a file between invocations so a script can enrol +//! once and then make several signed requests, the way a session on a phone +//! would. + +use std::io::{Read, Write}; +use std::net::TcpStream; +use std::path::{Path, PathBuf}; +use std::sync::Arc; + +use anyhow::{bail, Context, Result}; +use base64::Engine as _; +use clap::{Parser, Subcommand}; +use enclave_runtime::SoftwareAuthenticator; + +#[derive(Parser, Debug)] +#[command( + about = "Register a software passkey and make signed requests", + version +)] +struct Cli { + /// Base URL, e.g. `https://127.0.0.1:8443`. + #[arg(long)] + url: String, + + /// Where the passkey is kept between invocations. + #[arg(long, default_value = "/tmp/passkey-client.json")] + state: PathBuf, + + /// The relying-party id the enclave was configured with. + #[arg(long, default_value = "enclave.test")] + rp_id: String, + + /// The origin assertions must claim. Defaults to `https://`. + #[arg(long)] + origin: Option, + + /// Write each response's proof into this directory: `document.b64`, the + /// `certificate.der` this connection was served, and the `nonce.hex` the + /// request sent. Three files is what `nitro-attest --document + /// --peer-certificate --nonce` needs to check the binding for a request + /// this client signed — which it cannot make for itself. + #[arg(long)] + dump_proof: Option, + + /// Required PCR0, hex. Pins which runtime image is answering. + /// + /// The hypervisor measures this from the image and locks it, so nothing + /// running inside can choose it — which is what makes it the measurement + /// the other one rests on. It does not cover the guest, which is not in the + /// image. Required unless `--unsigned-emulator`. + #[arg(long)] + pcr0: Option, + + /// Required PCR16, hex. Pins which guest the runtime is running. + /// + /// The runtime extends this register with the guest's hash and locks it + /// before it can obtain a key, so it is the measurement that says which + /// application holds your data. It says that only beside `--pcr0`, because + /// the runtime is what writes it. Required — or `--guest` — unless + /// `--unsigned-emulator`. + #[arg(long)] + pcr16: Option, + + /// The guest component itself. Computes the PCR16 to require, and checks + /// the document's second hash as well. + /// + /// Never a substitute for `--pcr0`, for the same reason `--pcr16` is not: + /// both halves of what it checks are written by the runtime being attested. + #[arg(long)] + guest: Option, + + /// Trust root, PEM or DER. Defaults to the embedded AWS Nitro root. + #[arg(long)] + trust_root: Option, + + /// Accept a document that does not chain to the AWS root. + #[arg(long)] + allow_untrusted_root: bool, + + /// Reject a document older than this many seconds. + #[arg(long, default_value_t = 300)] + max_age: u64, + + /// Check the document's contents without verifying any signature. + /// + /// Only one producer needs this and it is not a real enclave: QEMU's + /// emulated NSM does not sign at all. With this on, **nothing establishes + /// who produced the document** — the nonce and certificate binding are + /// still checked, and they are this runtime's work, but the signature is + /// AWS hardware's and there is none to check. + /// + /// Never use it against anything you are trusting with a key. + #[arg(long)] + unsigned_emulator: bool, + + #[command(subcommand)] + command: Command, +} + +#[derive(Subcommand, Debug)] +enum Command { + /// Register a new passkey, which creates a new tenant. + Enrol, + /// A signed GET. + Get { + #[arg(long)] + path: String, + }, + /// A signed POST. + Post { + #[arg(long)] + path: String, + #[arg(long, default_value = "")] + body: String, + }, + /// Spend a token issued for one route against a different one. + /// + /// The harness uses this to show that an approval names an interaction and + /// cannot be moved off it — one no ordinary client can exercise, because a + /// client has no reason to build a request its own approval does not match. + /// + /// Note what this deliberately no longer demonstrates: a substituted + /// *body*. A token commits to no bytes, so sending different ones at the + /// approved route is not a refusal and there would be nothing to show. + Substitute { + /// The route the token is issued for. + #[arg(long)] + approved: String, + /// The route it is actually spent on. + #[arg(long)] + sent: String, + #[arg(long, default_value = "")] + body: String, + }, +} + +fn b64() -> base64::engine::general_purpose::GeneralPurpose { + base64::engine::general_purpose::URL_SAFE_NO_PAD +} + +fn main() -> Result<()> { + let cli = Cli::parse(); + let origin = cli + .origin + .clone() + .unwrap_or_else(|| format!("https://{}", cli.rp_id)); + + match &cli.command { + Command::Enrol => { + let authenticator = SoftwareAuthenticator::new(&cli.rp_id); + enrol(&cli, &origin, &authenticator)?; + save(&cli, &authenticator)?; + println!("enrolled"); + } + Command::Get { path } => { + let authenticator = load(&cli)?; + let (status, body) = signed(&cli, &origin, &authenticator, "GET", path, "", None)?; + save(&cli, &authenticator)?; + print!("{body}"); + std::process::exit(if (200..300).contains(&status) { 0 } else { 1 }); + } + Command::Post { path, body } => { + let authenticator = load(&cli)?; + let (status, response) = + signed(&cli, &origin, &authenticator, "POST", path, body, None)?; + save(&cli, &authenticator)?; + print!("{response}"); + std::process::exit(if (200..300).contains(&status) { 0 } else { 1 }); + } + Command::Substitute { + approved, + sent, + body, + } => { + let authenticator = load(&cli)?; + let (status, response) = signed( + &cli, + &origin, + &authenticator, + "POST", + sent, + body, + Some(approved), + )?; + save(&cli, &authenticator)?; + println!("{status}"); + print!("{response}"); + } + } + Ok(()) +} + +/// Just enough of the authenticator to rebuild it next time. +#[derive(serde::Serialize, serde::Deserialize)] +struct Saved { + credential_id: String, + key_pkcs8: String, + /// Carried between invocations so the counter keeps rising. A real + /// authenticator holds this in its own hardware; a shell script invoking a + /// fresh process per request has to write it down. + counter: u32, +} + +impl Saved { + fn of(a: &SoftwareAuthenticator) -> Self { + Saved { + credential_id: b64().encode(a.credential_id()), + key_pkcs8: b64().encode(a.private_key_pkcs8()), + counter: a.counter(), + } + } +} + +fn load(cli: &Cli) -> Result { + let raw = std::fs::read(&cli.state) + .with_context(|| format!("reading {} — run `enrol` first", cli.state.display()))?; + let saved: Saved = serde_json::from_slice(&raw).context("parsing saved passkey")?; + SoftwareAuthenticator::restore( + &cli.rp_id, + &b64().decode(saved.credential_id)?, + &b64().decode(saved.key_pkcs8)?, + saved.counter, + ) +} + +/// Write the passkey back, so the next invocation continues its counter. +fn save(cli: &Cli, a: &SoftwareAuthenticator) -> Result<()> { + std::fs::write(&cli.state, serde_json::to_vec(&Saved::of(a))?) + .with_context(|| format!("writing {}", cli.state.display())) +} + +fn enrol(cli: &Cli, origin: &str, a: &SoftwareAuthenticator) -> Result<()> { + let (options, opened_on) = + post_json(cli, "/auth/register/options", &serde_json::json!({}), None)?; + let opened_on = opened_on.expect("an unpinned exchange is always checked"); + let challenge = options["options"]["publicKey"]["challenge"] + .as_str() + .context("registration options carry no challenge")?; + let (verified, _) = post_json( + cli, + "/auth/register/verify", + &serde_json::json!({ + "registration_id": options["registration_id"], + "credential": a.register(challenge, origin), + }), + // The credential being registered is as worth protecting as an + // assertion, and for the same reason. + Some(&opened_on.certificate), + )?; + if verified["tenant_id"].as_str().is_none() { + bail!("registration was refused: {verified}"); + } + Ok(()) +} + +/// Ask for a challenge bound to a request, sign it, and send the request. +/// +/// `approved_route` exists only so the harness can take a token for one route +/// and spend it on another. +fn signed( + cli: &Cli, + origin: &str, + a: &SoftwareAuthenticator, + method: &str, + path: &str, + body: &str, + approved_route: Option<&str>, +) -> Result<(u16, String)> { + // The route the approval is *for*, which is normally the one being asked. + let approved = approved_route.unwrap_or(path); + let (approved_path, approved_query) = match approved.split_once('?') { + Some((p, q)) => (p, Some(q)), + None => (approved, None), + }; + // The challenge request is the one that goes first on a connection nothing + // has vouched for yet, and it is chosen for that: it carries the operation's + // hash but no approval, so a party that intercepted it would learn what is + // intended and hold nothing it could act on. + // + // Its response is attested like any other, which is what makes a separate + // probe route unnecessary — the round trip was going to happen anyway. + let (options, opened_on) = post_json( + cli, + "/auth/request/options", + &serde_json::json!({ + "credential_id": b64().encode(a.credential_id()), + "method": method, + "path": approved_path, + "query": approved_query, + }), + // Nothing to pin to yet: this is the exchange that establishes it, and + // it is chosen to go first precisely because it carries no approval. + None, + )?; + let opened_on = opened_on.expect("an unpinned exchange is always checked"); + + // Only now, with the far end identified, is it reasonable to ask a person + // to approve anything. `post_json` has already refused if the exchange + // could not be tied to this connection, so reaching here means it was. + let challenge = options["options"]["publicKey"]["challenge"] + .as_str() + .with_context(|| format!("challenge options carry no challenge: {options}"))?; + let assertion = a.assert(challenge, origin); + + // Trip two: hand the assertion back and take a token for the interaction. + // The token is what the interaction presents, so this is where the passkey + // ceremony ends and nothing after it needs the authenticator again — which + // is the whole point when the interaction is a stream with no single + // request to hang an assertion on. + let (granted, _) = post_json( + cli, + "/auth/request/verify", + &serde_json::json!({ + "challenge_id": options["challenge_id"] + .as_str() + .context("no challenge id")?, + "assertion": b64().encode(assertion.to_string()), + }), + // Checked after the handshake and *before* the assertion is written, so + // a changed far end means the credential is never sent. + Some(&opened_on.certificate), + )?; + let token = granted["token"] + .as_str() + .context("the runtime issued no token")?; + + // Trip three, pinned to the certificate the challenge exchange identified. + // The pin is checked after the handshake and *before* a byte goes out, so a + // changed far end means the interaction is never sent rather than sent and + // regretted. + // + // Nothing is retried on a fresh connection. The token is good once, so a + // retry would reach a party that has just failed to be the one approved, + // holding an approval that may already have been spent. + let (status, response, _) = request( + cli, + method, + path, + &[("authorization", format!("Bearer {token}"))], + body, + Some(&opened_on.certificate), + )?; + + Ok((status, response)) +} + +/// A JSON POST, optionally to a connection that must present `pinned`. +/// +/// The pin matters most on the exchange that carries the assertion. That +/// exchange used to open an unpinned connection, which meant the credential +/// went out to whoever answered: an interceptor could take it, redeem it at the +/// real enclave, and keep the token. Checking the response afterwards proves +/// only that the theft succeeded. +fn post_json( + cli: &Cli, + path: &str, + value: &serde_json::Value, + pinned: Option<&[u8]>, +) -> Result<(serde_json::Value, Option)> { + let (_, body, proof) = request( + cli, + "POST", + path, + &[("content-type", "application/json".to_string())], + &value.to_string(), + pinned, + )?; + let value = serde_json::from_str(&body) + .with_context(|| format!("{path} did not answer JSON: {body}"))?; + Ok((value, proof)) +} + +/// One HTTPS request, accepting whatever certificate is presented. +/// +/// The certificate is not what establishes trust — nothing vouches for it, and +/// nothing needs to. The attestation does, by naming the certificate this +/// connection presented, which is checked here before the response is returned. +/// What is *not* checked here is the signature and the chain: that needs a +/// trust root and expected measurements this client is not given, and +/// `nitro-attest` is the tool for it. +fn request( + cli: &Cli, + method: &str, + path: &str, + headers: &[(&str, String)], + body: &str, + // The certificate this connection must present, once one is known. `None` + // on the exchange that establishes it — there is nothing to pin to yet, + // which is why that exchange carries no approval. + pinned: Option<&[u8]>, +) -> Result<(u16, String, Option)> { + let authority = cli + .url + .trim_start_matches("https://") + .trim_end_matches('/') + .to_string(); + let config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions()? + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAny)) + .with_no_client_auth(); + + let name = rustls::pki_types::ServerName::try_from(cli.rp_id.clone())?; + let mut connection = rustls::ClientConnection::new(Arc::new(config), name)?; + let mut socket = + TcpStream::connect(&authority).with_context(|| format!("connecting to {authority}"))?; + + // Driven to completion here rather than left to the first write, because + // what the far end presents has to be known *before* anything is sent to + // it. TLS proves the peer holds the private key for the certificate it + // shows; the certificate itself is public and proves nothing on its own. + while connection.is_handshaking() { + connection + .complete_io(&mut socket) + .context("completing the TLS handshake")?; + } + let presented = connection + .peer_certificates() + .and_then(|c| c.first()) + .context("the server presented no certificate")? + .to_vec(); + + // The operation goes nowhere if this is not the enclave the approval was + // given to. Checked before a byte is written, so a changed far end means + // the request is never sent rather than sent and regretted. + if let Some(expected) = pinned { + anyhow::ensure!( + presented == expected, + "this connection presents a different certificate than the one the \ + approval was given to; the operation was not sent" + ); + } + + let mut tls = rustls::Stream::new(&mut connection, &mut socket); + + // Every request carries a nonce. The runtime refuses without one whether or + // not it attests, so a client that omits it reaches nothing. + let nonce = fresh_nonce(); + let mut request = format!( + "{method} {path} HTTP/1.1\r\nHost: {}\r\nConnection: close\r\n\ + x-enclave-nonce: {}\r\nContent-Length: {}\r\n", + cli.rp_id, + encode_nonce(&nonce), + body.len() + ); + for (name, value) in headers { + request.push_str(&format!("{name}: {value}\r\n")); + } + request.push_str("\r\n"); + request.push_str(body); + tls.write_all(request.as_bytes())?; + + let mut raw = Vec::new(); + let _ = tls.read_to_end(&mut raw); + let split = raw + .windows(4) + .position(|w| w == b"\r\n\r\n") + .context("no response headers")?; + let head = String::from_utf8_lossy(&raw[..split]).to_string(); + let status: u16 = head + .lines() + .next() + .and_then(|l| l.split_whitespace().nth(1)) + .context("no status line")? + .parse()?; + + // On a pinned connection there is nothing left to establish: the far end + // presented the certificate the approval was given to, and TLS proved it + // holds the private key. The runtime does not attest these responses for + // exactly that reason, so there is no document here to read. + // + // On an unpinned one there is everything to establish, and it is checked + // before the body is even looked at. A response this client cannot tie to + // the enclave on the other end is not a response it will act on, and a + // deployment with attestation off is one it will not talk to — nothing to + // check means nothing it can promise. + let proof = match pinned { + // A pinned response carries no document by design — the runtime attests + // the `/auth/` exchange and leaves the rest alone — so there is nothing + // here to check and nothing to dump. Dumping unconditionally used to + // fail *after* the interaction had already run. + Some(_) => None, + None => Some( + connection_proof(&head, &nonce, &presented, &TrustConfig::from(cli)?) + .with_context(|| format!("{method} {path} could not be tied to this connection"))?, + ), + }; + + // Only the attested exchange has a proof worth keeping, and it is the one a + // verifier needs: the document, the certificate it binds, and the nonce it + // quotes. The interaction that follows is covered by the pin. + if proof.is_some() { + if let Some(dir) = &cli.dump_proof { + dump_proof(dir, &head, &nonce, &connection)?; + } + } + + let rest = &raw[split + 4..]; + let body = if head + .to_ascii_lowercase() + .contains("transfer-encoding: chunked") + { + dechunk(rest) + } else { + rest.to_vec() + }; + Ok((status, String::from_utf8_lossy(&body).to_string(), proof)) +} + +/// A fresh nonce for one request. +/// +/// From the OS, because a nonce a third party can predict is not a nonce — and +/// this one is checked: the document that comes back must quote it, or the +/// response is refused before its body is read. +fn fresh_nonce() -> [u8; 20] { + let mut nonce = [0u8; 20]; + getrandom::fill(&mut nonce).expect("the OS has entropy"); + nonce +} + +fn encode_nonce(nonce: &[u8]) -> String { + use base64::Engine as _; + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(nonce) +} + +/// The three things a verifier needs and cannot recover after the fact: what +/// the enclave signed, which certificate this connection was actually served, +/// and which nonce was asked for. +/// What one response proves about the connection it arrived on. +/// +/// The document is generated before the request is routed, so it says nothing +/// about the body below it and nothing about whether the guest ran. What it +/// does say is that *this* enclave terminated *this* connection, just now — +/// and that is the claim the operation depends on. +#[derive(Debug, Clone, PartialEq, Eq)] +struct ConnectionProof { + /// The leaf this connection presented, which the document was checked to + /// name. Kept whole rather than hashed because the next connection is + /// pinned against it, and TLS compares certificates rather than digests. + certificate: Vec, + guest: [u8; 32], +} + +/// Verify a response's attestation against the connection it arrived on. +/// +/// Everything, in the order it has to happen: +/// +/// - the **signature and certificate chain**, so the document came from a real +/// Nitro enclave and not from whoever answered the socket; +/// - its **age**, so it was made now rather than captured earlier; +/// - the **nonce**, so it was made for this caller and not replayed from +/// somebody else's exchange; +/// - the **certificate** it binds, against the one this connection actually +/// presented, so the enclave that signed it is the party on the other end +/// rather than one being relayed by something in between; +/// - and, when given, **PCR0** and the guest hash — *which* enclave and *which* +/// application, not merely that it is some enclave. +/// +/// The chain check is what the other four rest on. Without it a party that +/// terminated TLS could mint a document naming its own certificate and quoting +/// the nonce, and every remaining check would pass. +fn connection_proof( + head: &str, + nonce: &[u8], + presented: &[u8], + trust: &TrustConfig, +) -> Result { + let encoded = head + .lines() + .find(|l| l.to_ascii_lowercase().starts_with("x-enclave-attestation:")) + .and_then(|l| l.split_once(':')) + .map(|(_, v)| v.trim()) + .context("the response carried no x-enclave-attestation header")?; + let cose = base64::engine::general_purpose::STANDARD + .decode(encoded) + .context("the attestation header is not base64")?; + + let now = std::time::SystemTime::now(); + let verified = if trust.unsigned_emulator { + nitro_attestation::Verified { + document: nitro_attestation::parse(&cose).context("parsing the document")?, + trust: nitro_attestation::Trust::Unsigned, + } + } else { + nitro_attestation::verify( + &cose, + &nitro_attestation::VerifyOptions { + trust_root: trust.root.clone(), + now, + allow_untrusted_root: trust.allow_untrusted_root, + }, + ) + .context("the document did not verify")? + }; + + // What the runtime binds: the leaf this connection was served, and the + // guest component behind it. The certificate half is checked always; the + // guest half only when the caller supplied the component to check against. + let hashes = nitro_attestation::AttestationHashes::parse( + verified + .document + .user_data + .as_deref() + .context("the document binds no user_data")?, + ) + .context("the document's user_data is not the runtime's binding")?; + anyhow::ensure!( + hashes.tls_certificate == nitro_attestation::sha256(presented), + "the document binds a certificate this connection was never served: \ + the enclave that signed it is not the party answering here" + ); + + let mut expectations = nitro_attestation::Expectations { + nonce: Some(nonce.to_vec()), + max_age: Some(trust.max_age), + ..Default::default() + }; + if let Some(pcr0) = &trust.pcr0 { + expectations = expectations.pcr0(pcr0.clone()); + } + // Missing from the document is a refusal, not a skipped check: a runtime + // that never locked the register attests no guest at all. + if let Some(pcr16) = &trust.pcr16 { + expectations = expectations.pcr(nitro_attestation::PCR_GUEST, pcr16.clone()); + } + if let Some(guest) = &trust.guest { + expectations.user_data = + Some(nitro_attestation::AttestationHashes::new(presented, guest).serialize()); + } + verified + .expect(&expectations, now) + .context("the document did not meet expectations")?; + + Ok(ConnectionProof { + certificate: presented.to_vec(), + guest: hashes.guest, + }) +} + +/// What the client requires of a document before it acts on the response. +struct TrustConfig { + root: Vec, + allow_untrusted_root: bool, + unsigned_emulator: bool, + max_age: std::time::Duration, + pcr0: Option>, + /// The guest register to require, from `--pcr16` or computed from + /// `--guest`. + pcr16: Option>, + guest: Option>, +} + +impl TrustConfig { + fn from(cli: &Cli) -> Result { + // A verified chain says "a genuine Nitro enclave". It does not say + // *which* one, and an attacker who can run their own gets a document + // that passes every other check here. + // + // **Two measurements close that, and neither stands in for the other.** + // PCR0 is measured by the hypervisor from the image and locked, so no + // software inside the enclave can choose it — but the image no longer + // contains the guest. PCR16 is the guest, extended and locked by the + // runtime before it could obtain a key — but because the runtime writes + // it, an enclave running an attacker's runtime can claim your guest's + // value while running nothing of the kind. PCR0 says the runtime that + // wrote PCR16 is yours; PCR16 says which application it loaded. + // + // `--guest` is how most callers supply PCR16, and it checks the + // document's guest hash as well. Both halves of that come from the + // runtime, so it is never a substitute for `--pcr0`. + // + // `--unsigned-emulator` is the one exception, and an explicit one: QEMU + // has no stable PCR0 to pin, and nothing there is signed anyway. + let guest = match &cli.guest { + Some(path) => { + Some(std::fs::read(path).with_context(|| format!("reading {}", path.display()))?) + } + None => None, + }; + let measured = guest + .as_deref() + .map(|component| nitro_attestation::guest_pcr(component).to_vec()); + let pcr16 = match (&cli.pcr16, measured) { + (Some(pinned), measured) => { + let pinned = hex::decode(pinned.trim()).context("--pcr16 is not hex")?; + if let Some(measured) = measured { + anyhow::ensure!( + measured == pinned, + "--pcr16 and --guest name different guests: the component measures \ + to {}", + hex::encode(&measured) + ); + } + Some(pinned) + } + (None, measured) => measured, + }; + + if !cli.unsigned_emulator { + if cli.pcr0.is_none() { + bail!( + "refusing to talk to an unidentified enclave: pass --pcr0, which is the \ + measurement the hypervisor takes of the image and the only one an \ + attacker's own enclave cannot claim. --guest and --pcr16 are required \ + alongside it, not instead of it: the runtime being attested is what \ + writes them. Against QEMU, pass --unsigned-emulator." + ); + } + if pcr16.is_none() { + bail!( + "refusing to talk to an enclave whose guest is unidentified: pass --pcr16, \ + or --guest with the component itself. PCR0 measures the runtime image, \ + and the guest is not in it — the runtime loads it at boot, measures it \ + into PCR16 and locks that register before it can obtain a key. PCR16 is \ + the measurement that says which application holds your data." + ); + } + } + Ok(TrustConfig { + root: match &cli.trust_root { + Some(path) => { + std::fs::read(path).with_context(|| format!("reading {}", path.display()))? + } + None => nitro_attestation::AWS_NITRO_ROOT_G1_PEM.as_bytes().to_vec(), + }, + allow_untrusted_root: cli.allow_untrusted_root, + unsigned_emulator: cli.unsigned_emulator, + max_age: std::time::Duration::from_secs(cli.max_age), + pcr0: match &cli.pcr0 { + Some(hex) => Some(hex::decode(hex.trim()).context("--pcr0 is not hex")?), + None => None, + }, + pcr16, + guest, + }) + } +} + +fn dump_proof( + dir: &Path, + head: &str, + nonce: &[u8], + connection: &rustls::ClientConnection, +) -> Result<()> { + let document = head + .lines() + .find(|l| l.to_ascii_lowercase().starts_with("x-enclave-attestation:")) + .and_then(|l| l.split_once(':')) + .map(|(_, v)| v.trim()) + .context("the response carried no x-enclave-attestation header")?; + let certificate = connection + .peer_certificates() + .and_then(|c| c.first()) + .context("the server presented no certificate")?; + + std::fs::create_dir_all(dir).with_context(|| format!("creating {}", dir.display()))?; + std::fs::write(dir.join("document.b64"), document)?; + std::fs::write(dir.join("certificate.der"), certificate.as_ref())?; + std::fs::write(dir.join("nonce.hex"), hex::encode(nonce))?; + Ok(()) +} + +fn dechunk(body: &[u8]) -> Vec { + let mut out = Vec::new(); + let mut rest = body; + while let Some(end) = rest.windows(2).position(|w| w == b"\r\n") { + let header = String::from_utf8_lossy(&rest[..end]); + let Ok(size) = usize::from_str_radix(header.split(';').next().unwrap_or("").trim(), 16) + else { + break; + }; + rest = &rest[end + 2..]; + if size == 0 || rest.len() < size { + break; + } + out.extend_from_slice(&rest[..size]); + rest = rest.get(size + 2..).unwrap_or(&[]); + } + out +} + +#[derive(Debug)] +struct AcceptAny; + +impl rustls::client::danger::ServerCertVerifier for AcceptAny { + fn verify_server_cert( + &self, + _e: &rustls::pki_types::CertificateDer<'_>, + _i: &[rustls::pki_types::CertificateDer<'_>], + _s: &rustls::pki_types::ServerName<'_>, + _o: &[u8], + _n: rustls::pki_types::UnixTime, + ) -> Result { + Ok(rustls::client::danger::ServerCertVerified::assertion()) + } + + fn verify_tls12_signature( + &self, + _m: &[u8], + _c: &rustls::pki_types::CertificateDer<'_>, + _d: &rustls::DigitallySignedStruct, + ) -> Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + + fn verify_tls13_signature( + &self, + _m: &[u8], + _c: &rustls::pki_types::CertificateDer<'_>, + _d: &rustls::DigitallySignedStruct, + ) -> Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + + fn supported_verify_schemes(&self) -> Vec { + rustls::crypto::aws_lc_rs::default_provider() + .signature_verification_algorithms + .supported_schemes() + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nitro_attestation::testing::TestChain; + use nitro_attestation::AttestationHashes; + + /// What a client requires, pointed at a test chain rather than AWS. + /// + /// `allow_untrusted_root` stays **false**: it does not mean "this root is + /// not AWS", it means "accept a document that chains to nothing I pinned", + /// which would make every test below vacuous. Pinning the test chain's own + /// root is what keeps the chain check real. + fn trust(chain: &TestChain) -> TrustConfig { + TrustConfig { + root: chain.root_der().to_vec(), + allow_untrusted_root: false, + unsigned_emulator: false, + max_age: std::time::Duration::from_secs(300), + pcr0: None, + pcr16: None, + guest: None, + } + } + + const GUEST: &[u8] = b"a-guest"; + + /// The registers an enclave running [`GUEST`] has locked: its image, and + /// the guest it measured. + fn registers() -> std::collections::BTreeMap> { + [ + (0u32, vec![0x5a; 48]), + ( + nitro_attestation::PCR_GUEST, + nitro_attestation::guest_pcr(GUEST).to_vec(), + ), + ] + .into() + } + + /// A response head carrying a document built for `cert` and `nonce`. + fn head_for(chain: &TestChain, cert: &[u8], nonce: &[u8]) -> String { + head_with(chain, cert, nonce, registers()) + } + + /// The same, listing exactly the registers given. + fn head_with( + chain: &TestChain, + cert: &[u8], + nonce: &[u8], + pcrs: std::collections::BTreeMap>, + ) -> String { + let user_data = AttestationHashes::new(cert, GUEST).serialize(); + let cose = chain + .document_with_pcrs(Some(user_data), Some(nonce.to_vec()), pcrs) + .expect("a document"); + format!( + "HTTP/1.1 200 OK\r\nx-enclave-attestation: {}\r\n", + base64::engine::general_purpose::STANDARD.encode(cose) + ) + } + + #[test] + fn a_document_for_this_connection_and_this_nonce_is_accepted() { + let chain = TestChain::new().expect("chain"); + let cert = b"the-leaf-this-connection-presented"; + let nonce = [7u8; 20]; + let proof = connection_proof( + &head_for(&chain, cert, &nonce), + &nonce, + cert, + &trust(&chain), + ) + .expect("the document names this certificate and quotes this nonce"); + assert_eq!(proof.certificate, cert.to_vec()); + } + + /// **The forgery.** A document nobody signed, naming the forger's own + /// certificate and quoting the nonce it was just sent. + /// + /// Every check except the chain passes: the nonce matches, and the + /// certificate it binds really is the one this connection presented — + /// because the forger chose both. Only verifying the signature and the + /// chain catches it, which is why parsing the document was never enough. + #[test] + fn a_document_that_nobody_signed_is_refused() { + let forger = TestChain::new().expect("the forger's own chain"); + let honest = TestChain::new().expect("the enclave's chain"); + let cert = b"the-forgers-leaf"; + let nonce = [7u8; 20]; + + // Signed by a chain the client does not trust. + let err = connection_proof( + &head_for(&forger, cert, &nonce), + &nonce, + cert, + &trust(&honest), + ) + .expect_err("a document from an untrusted chain was accepted"); + assert!( + format!("{err:#}").contains("did not verify"), + "the refusal did not name the reason: {err:#}" + ); + } + + /// **The relay.** A genuine document from the real enclave, replayed by + /// something that terminated TLS itself. + #[test] + fn a_document_naming_another_certificate_is_refused() { + let chain = TestChain::new().expect("chain"); + let nonce = [7u8; 20]; + let head = head_for(&chain, b"the-real-enclaves-leaf", &nonce); + let err = connection_proof(&head, &nonce, b"the-interceptors-leaf", &trust(&chain)) + .expect_err("a relayed document was accepted"); + assert!( + format!("{err:#}").contains("never served"), + "the refusal did not name the reason: {err:#}" + ); + } + + /// **The replay.** A document made earlier, for an earlier request. + #[test] + fn a_document_quoting_another_nonce_is_refused() { + let chain = TestChain::new().expect("chain"); + let cert = b"the-leaf"; + let head = head_for(&chain, cert, &[1u8; 20]); + assert!( + connection_proof(&head, &[2u8; 20], cert, &trust(&chain)).is_err(), + "a document made for another request was accepted" + ); + } + + /// A document older than the client will accept. + #[test] + fn a_stale_document_is_refused() { + let chain = TestChain::new().expect("chain"); + let cert = b"the-leaf"; + let nonce = [4u8; 20]; + let mut trust = trust(&chain); + trust.max_age = std::time::Duration::from_nanos(1); + std::thread::sleep(std::time::Duration::from_millis(5)); + assert!( + connection_proof(&head_for(&chain, cert, &nonce), &nonce, cert, &trust).is_err(), + "a document older than the client's window was accepted" + ); + } + + /// The wrong image, caught by PCR0 when the caller pins one. + #[test] + fn a_document_from_another_image_is_refused() { + let chain = TestChain::new().expect("chain"); + let cert = b"the-leaf"; + let nonce = [5u8; 20]; + let mut trust = trust(&chain); + trust.pcr0 = Some(vec![0xff; 48]); + assert!( + connection_proof(&head_for(&chain, cert, &nonce), &nonce, cert, &trust).is_err(), + "a document from an unexpected image was accepted" + ); + } + + /// Fail closed: a deployment that does not attest is one this client will + /// not act on, because there is nothing for it to check. + #[test] + fn a_response_without_a_document_is_refused() { + let chain = TestChain::new().expect("chain"); + let err = connection_proof("HTTP/1.1 200 OK\r\n", &[0u8; 20], b"leaf", &trust(&chain)) + .expect_err("an unattested response was accepted"); + assert!( + format!("{err:#}").contains("no x-enclave-attestation"), + "{err:#}" + ); + } + + /// The continuity check, which is what pins the interaction's connection. + #[test] + fn two_exchanges_on_different_certificates_do_not_match() { + let chain = TestChain::new().expect("chain"); + let nonce = [3u8; 20]; + let t = trust(&chain); + + let opened = connection_proof( + &head_for(&chain, b"leaf-one", &nonce), + &nonce, + b"leaf-one", + &t, + ) + .expect("the first exchange"); + let answered = connection_proof( + &head_for(&chain, b"leaf-two", &nonce), + &nonce, + b"leaf-two", + &t, + ) + .expect("the second exchange"); + assert_ne!( + opened.certificate, answered.certificate, + "a changed certificate must be visible, or the pin cannot fire" + ); + + let again = connection_proof( + &head_for(&chain, b"leaf-one", &nonce), + &nonce, + b"leaf-one", + &t, + ) + .expect("the same enclave again"); + assert_eq!(opened.certificate, again.certificate); + } + + /// **A chain alone is not an identity.** + /// + /// Verifying the signature says a genuine Nitro enclave answered. It does + /// not say whose. Refusing to run without measurements is what stops the + /// default from reading as more safety than it gives. + #[test] + fn a_client_refuses_to_run_without_expected_measurements() { + let cli = Cli::parse_from([ + "passkey-client", + "--url", + "https://e.test", + "get", + "--path", + "/", + ]); + let err = match TrustConfig::from(&cli) { + Err(e) => e, + Ok(_) => panic!("an unidentified enclave was accepted"), + }; + assert!( + format!("{err:#}").contains("unidentified enclave"), + "{err:#}" + ); + } + + fn cli_with(flags: &[&str]) -> Cli { + let mut args = vec!["passkey-client", "--url", "https://e.test"]; + args.extend_from_slice(flags); + args.extend_from_slice(&["get", "--path", "/"]); + Cli::parse_from(args) + } + + fn guest_file(name: &str, bytes: &[u8]) -> PathBuf { + let path = + std::env::temp_dir().join(format!("passkey-client-{}-{name}.wasm", std::process::id())); + std::fs::write(&path, bytes).expect("writing a guest file"); + path + } + + fn refusal(cli: &Cli) -> String { + match TrustConfig::from(cli) { + Err(e) => format!("{e:#}"), + Ok(_) => panic!("an unidentified enclave was accepted"), + } + } + + /// **`--guest` alone is not an identity.** + /// + /// It yields a PCR16 and a `user_data` guest hash, and the runtime being + /// attested writes both — so an attacker running their own runtime signs a + /// genuine document claiming whatever guest you asked for. Only PCR0, + /// measured by the hypervisor and locked, says whose runtime wrote them. + #[test] + fn a_guest_alone_is_not_an_identity() { + let text = refusal(&cli_with(&["--guest", "/dev/null"])); + assert!(text.contains("--pcr0"), "{text}"); + } + + /// **PCR0 alone no longer identifies the guest.** The image does not + /// contain it, so a client pinning only the runtime would accept that + /// runtime serving any application at all. + #[test] + fn pinning_the_runtime_alone_does_not_identify_the_guest() { + let text = refusal(&cli_with(&["--pcr0", &"ab".repeat(48)])); + assert!(text.contains("PCR16"), "{text}"); + } + + #[test] + fn pinning_both_measurements_identifies_the_enclave() { + let trust = TrustConfig::from(&cli_with(&[ + "--pcr0", + &"ab".repeat(48), + "--pcr16", + &"cd".repeat(48), + ])) + .expect("both measurements were not accepted"); + assert_eq!(trust.pcr16, Some(vec![0xcd; 48])); + } + + /// The component itself is the usual way to supply PCR16, and it becomes + /// exactly the value the runtime would have locked. + #[test] + fn a_guest_file_becomes_the_register_it_measures_to() { + let path = guest_file("measured", GUEST); + let trust = TrustConfig::from(&cli_with(&[ + "--pcr0", + &"ab".repeat(48), + "--guest", + path.to_str().unwrap(), + ])); + let _ = std::fs::remove_file(&path); + + let trust = trust.expect("a guest file was not accepted"); + assert_eq!( + trust.pcr16, + Some(nitro_attestation::guest_pcr(GUEST).to_vec()) + ); + assert_eq!(trust.guest.as_deref(), Some(GUEST)); + } + + #[test] + fn a_pcr16_and_a_guest_file_that_disagree_are_refused() { + let path = guest_file("disagreeing", GUEST); + let cli = cli_with(&[ + "--pcr0", + &"ab".repeat(48), + "--pcr16", + &"00".repeat(48), + "--guest", + path.to_str().unwrap(), + ]); + let text = refusal(&cli); + let _ = std::fs::remove_file(&path); + assert!(text.contains("different guests"), "{text}"); + } + + /// **Another guest**, behind the right runtime. + #[test] + fn a_document_from_another_guest_is_refused() { + let chain = TestChain::new().expect("chain"); + let cert = b"the-leaf"; + let nonce = [6u8; 20]; + let mut trust = trust(&chain); + trust.pcr0 = Some(vec![0x5a; 48]); + trust.pcr16 = Some(nitro_attestation::guest_pcr(b"the approved guest").to_vec()); + let err = connection_proof(&head_for(&chain, cert, &nonce), &nonce, cert, &trust) + .expect_err("a document for another guest was accepted"); + assert!(format!("{err:#}").contains("PCR16 mismatch"), "{err:#}"); + } + + /// **No guest at all.** A runtime that never locked the register attests + /// documents without it, and a client pinning a guest refuses that rather + /// than skipping the check. + #[test] + fn a_document_that_measured_no_guest_is_refused() { + let chain = TestChain::new().expect("chain"); + let cert = b"the-leaf"; + let nonce = [7u8; 20]; + let mut trust = trust(&chain); + trust.pcr16 = Some(nitro_attestation::guest_pcr(GUEST).to_vec()); + let head = head_with(&chain, cert, &nonce, [(0u32, vec![0x5a; 48])].into()); + let err = connection_proof(&head, &nonce, cert, &trust) + .expect_err("a document with no guest register was accepted"); + assert!(format!("{err:#}").contains("no PCR16"), "{err:#}"); + } + + #[test] + fn a_document_for_the_pinned_runtime_and_guest_is_accepted() { + let chain = TestChain::new().expect("chain"); + let cert = b"the-leaf"; + let nonce = [8u8; 20]; + let mut trust = trust(&chain); + trust.pcr0 = Some(vec![0x5a; 48]); + trust.pcr16 = Some(nitro_attestation::guest_pcr(GUEST).to_vec()); + trust.guest = Some(GUEST.to_vec()); + connection_proof(&head_for(&chain, cert, &nonce), &nonce, cert, &trust) + .expect("the pinned runtime and guest were refused"); + } + + /// The exception is explicit, and only for the one producer that needs it. + #[test] + fn the_emulator_exception_must_be_asked_for_by_name() { + let trust = + TrustConfig::from(&cli_with(&["--unsigned-emulator"])).expect("the emulator exception"); + assert!(trust.unsigned_emulator); + assert!( + trust.pcr0.is_none() && trust.pcr16.is_none(), + "the exception should not invent a measurement" + ); + } +} diff --git a/runtime/src/boot.rs b/runtime/src/boot.rs new file mode 100644 index 0000000..47863d2 --- /dev/null +++ b/runtime/src/boot.rs @@ -0,0 +1,950 @@ +//! Deciding whether this enclave is entitled to the state it is about to load. +//! +//! `Store::open` used to end with `None => Store::format(…)`: an enclave +//! pointed at an empty store made a fresh filesystem and served it. It was +//! correctly signed, correctly hash-chained, and completely wrong — every +//! check the design makes passed, because they all attested to the *new* +//! filesystem while the guest saw an empty database where its data should have +//! been. +//! +//! `store/root.rs` already documents the neighbouring risk, rollback: *"a cold +//! mount cannot distinguish 'the tip is N' from 'the tip is N, and the store is +//! hiding N+1'"*. This is the worse one it does not name — a cold mount cannot +//! distinguish "this filesystem is new" from "everything has been hidden". +//! +//! ## The receipt +//! +//! Genesis writes a **state-origin receipt**: an NSM attestation document +//! whose `user_data` commits to a hash over this filesystem's identity. The +//! host cannot forge one, because AWS signs it. Following +//! [ArkLabsHQ/enclave#151](https://github.com/ArkLabsHQ/enclave/pull/151), +//! whose `runtime/boot.go` puts it exactly right: +//! +//! > Committing to it in an NSM attestation is what lets a later boot — or a +//! > successor across a migration — prove the state it loaded is the state +//! > some enclave of a known PCR0 actually wrote, rather than something the +//! > host substituted. +//! +//! ## Three things make a missing receipt mean something +//! +//! The receipt is self-authenticating, so the host cannot forge one. It can +//! still *delete* or *hide* one, and "no receipt" is what authorises genesis — +//! so the absence has to be as trustworthy as the presence: +//! +//! - **Object Lock COMPLIANCE** on the roots bucket: nobody, including the +//! account root, can delete the receipt once written. +//! - **Attested bucket identity**: the bucket names are baked into the enclave +//! image, so PCR0 covers them. Without this the host simply points the +//! enclave at an empty bucket and every other check passes. +//! - **TLS to S3, validated inside the enclave**: the parent proxies the +//! bytes but cannot substitute them. It can block, which fails closed. +//! +//! What remains out of scope, unchanged from `root.rs`: S3 lying about `HEAD`. +//! That is AWS, whom we already trust for the signature on the receipt itself. +//! +//! ## Which code may hold the state +//! +//! Not this module's decision. An enclave that gets past opening the key was +//! released it by KMS, against a policy naming one runtime image (PCR0) and one +//! guest (PCR16), and changing either is an edit to that policy by whoever +//! controls the key. A boot machine that also refused pairs would be a second +//! copy of that decision, able only to disagree with the real one. +//! +//! What boot adds is a **record**. The first time a runtime and guest hold this +//! state, the enclave attests a *pair record* — a document carrying its PCR0 +//! and PCR16 and committing to this `state_root` — and stores it undeletably +//! under a key derived from the pair. Later boots of that pair find it and +//! write nothing. The store therefore holds one signed record for each distinct +//! pair that has ever held the state. +//! +//! It records *which* pairs, not *in what order*: returning to an earlier pair +//! finds that pair's record and adds nothing. The order approvals were given in +//! lives where they were given, in CloudTrail's record of key-policy edits. +//! +//! Under the static development key source there is no policy at all, so +//! nothing decides which pair may boot. That source protects nothing, and the +//! image environment that selects it is measured. + +use std::sync::Arc; + +use anyhow::{bail, Context, Result}; +use nitro_attestation::{Expectations, Trust, VerifyOptions, AWS_NITRO_ROOT_G1_PEM}; +use nitro_nsm::{AttestationRequest, Nsm, PCR_GUEST, PCR_ZERO}; +use s3fs_core::backend::{Backend, ObjectLock, PutBlobInput}; +use s3fs_core::{FsError, MasterSecret}; + +use crate::keys::{MasterKeySource, SealedKey}; +use crate::mount::{Backends, MountConfig, Mounted}; + +/// `user_data` purpose for the receipt genesis writes. +const PURPOSE_STATE_ORIGIN: &str = "s3fs-state-origin"; +/// `user_data` purpose for the record a runtime and guest leave the first time +/// they hold this state. +const PURPOSE_PAIR: &str = "s3fs-pair"; + +/// Schema string inside the `state_root` pre-image. Bump it and every existing +/// receipt stops verifying, which is the intended effect of changing what a +/// receipt means. +const STATE_ROOT_SCHEMA: &str = "s3fs/state-origin/v1"; + +/// Which boot this is. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BootMode { + /// No receipt and no filesystem: create both. + Genesis, + /// This runtime and guest have held this state before. + Resume, + /// The first boot of this runtime and guest on this state: a new guest, a + /// new runtime image, or both. Recorded rather than refused — the key + /// policy is what allowed it. + /// + /// Derived from whether this pair's record exists, not by comparing with + /// whoever ran genesis, which would call every restart after an upgrade + /// another upgrade. + Upgrade, +} + +/// The two measurements that say what an enclave is running. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Pair { + /// The runtime image, measured by the hypervisor. + pub pcr0: [u8; 48], + /// The guest component, measured by that runtime into [`PCR_GUEST`] and + /// locked before it asked for a key. + pub pcr16: [u8; 48], +} + +impl Pair { + /// Read both from the device, refusing if the guest was never measured. + /// + /// The first thing [`boot`] does, before anything is read from the store or + /// asked of KMS. An enclave that reached KMS with PCR16 unlocked would + /// present an attestation with no guest in it, and could still extend the + /// register after being released a key. + pub fn read(nsm: &dyn Nsm) -> Result { + let pcr0 = nsm + .describe_pcr(0) + .context("reading PCR0; the boot machine cannot record what it cannot read")?; + let guest = nsm + .describe_pcr(PCR_GUEST) + .context("reading PCR16, where the guest is measured")?; + if !guest.locked { + bail!( + "PCR16 is not locked, so this enclave has not measured its guest. It must be \ + measured and locked before any key is asked for: an attestation carries only \ + locked registers, so a key policy pinning the guest would see none. Refusing \ + to boot." + ); + } + if guest.value == PCR_ZERO { + bail!( + "PCR16 is locked but was never extended, so it names no guest. Refusing to boot." + ); + } + Ok(Pair { + pcr0: register(&pcr0.value, 0)?, + pcr16: register(&guest.value, PCR_GUEST)?, + }) + } + + /// Where this pair's record lives. + /// + /// Derived, like the receipt and the sealed key, so "no record" is an + /// answer rather than a failure to look in the right place. Both registers + /// are SHA-384 and so always 48 bytes, which keeps the concatenation + /// unambiguous. + pub fn record_key(&self, prefix: &str, fs_uuid: &[u8; 16]) -> String { + let mut both = [0u8; 96]; + both[..48].copy_from_slice(&self.pcr0); + both[48..].copy_from_slice(&self.pcr16); + format!( + "{prefix}origin/{}.pair.{}", + hex::encode(fs_uuid), + hex::encode(nitro_attestation::sha256(&both)) + ) + } +} + +fn register(value: &[u8], index: u16) -> Result<[u8; 48]> { + value.try_into().with_context(|| { + format!( + "PCR{index} is {} bytes; a SHA-384 register is 48", + value.len() + ) + }) +} + +/// What a boot established, for logging and for the health endpoint. +#[derive(Debug)] +pub struct Booted { + pub mode: BootMode, + pub mounted: Mounted, + pub state_root: [u8; 32], + /// The runtime image and guest this enclave is running. + pub pair: Pair, + /// The recovered master secret. + /// + /// Returned rather than dropped because per-client filesystems derive + /// their key material from it, and there is nowhere else to get it: the + /// runtime's own `KeyMaterial` holds only what was derived *for* the + /// runtime filesystem. + /// + /// It zeroizes on drop and never prints. It is already resident in this + /// process either way; what this changes is that it stays reachable, and + /// the alternative was re-opening the key source per client. + pub master: MasterSecret, +} + +/// The identity a receipt commits to. +/// +/// Every field is something an attacker could otherwise vary while leaving the +/// rest of the design intact: the filesystem, the store it lives in, the +/// history it descends from, and the key that reads it. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct StateIdentity { + pub fs_uuid: [u8; 16], + pub data_bucket: String, + pub roots_bucket: String, + pub bucket_prefix: String, + /// Hash of the seq-0 record: *which* history, not merely which bucket. + pub genesis_root_hash: [u8; 32], + /// SHA-256 of the **sealed** key. Never the key: a receipt is readable by + /// anyone who can read the bucket it sits in. + pub sealed_key_sha256: [u8; 32], +} + +impl StateIdentity { + /// The value a receipt's `user_data` commits to. + /// + /// Deterministic: CBOR with fields in a fixed order, hashed. Two enclaves + /// looking at the same filesystem must compute the same 32 bytes or the + /// receipt is useless. + pub fn state_root(&self) -> [u8; 32] { + let value = ciborium::Value::Array(vec![ + ciborium::Value::Text(STATE_ROOT_SCHEMA.into()), + ciborium::Value::Bytes(self.fs_uuid.to_vec()), + ciborium::Value::Text(self.data_bucket.clone()), + ciborium::Value::Text(self.roots_bucket.clone()), + ciborium::Value::Text(self.bucket_prefix.clone()), + ciborium::Value::Bytes(self.genesis_root_hash.to_vec()), + ciborium::Value::Bytes(self.sealed_key_sha256.to_vec()), + ]); + let mut encoded = Vec::new(); + ciborium::into_writer(&value, &mut encoded).expect("writing to a Vec cannot fail"); + *blake3::hash(&encoded).as_bytes() + } +} + +/// `user_data` payload of a receipt or a pair record. +/// +/// Every receipt already written commits to this encoding, so changing it +/// would make every existing filesystem unmountable. +fn receipt_payload(purpose: &str, state_root: &[u8; 32]) -> Vec { + let fields = vec![ + ( + ciborium::Value::Text("purpose".into()), + ciborium::Value::Text(purpose.to_string()), + ), + ( + ciborium::Value::Text("state_root".into()), + ciborium::Value::Bytes(state_root.to_vec()), + ), + ]; + let mut out = Vec::new(); + ciborium::into_writer(&ciborium::Value::Map(fields), &mut out) + .expect("writing to a Vec cannot fail"); + out +} + +/// Object keys, derived from the filesystem id alone. +/// +/// Derivable without reading anything, which is what makes a 404 meaningful: +/// the enclave knows exactly where to look, so "not there" is an answer rather +/// than a failure to find the right place. +fn receipt_key(prefix: &str, fs_uuid: &[u8; 16]) -> String { + format!("{prefix}origin/{}.receipt", hex::encode(fs_uuid)) +} + +fn sealed_key_key(prefix: &str, fs_uuid: &[u8; 16]) -> String { + format!("{prefix}origin/{}.key", hex::encode(fs_uuid)) +} + +/// How a receipt is checked. QEMU's NSM does not sign, so the harness needs a +/// concession that production must never have. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ReceiptTrust { + /// Signature and chain to the AWS Nitro root. + Required, + /// Contents only, because the emulated NSM produces unsigned documents. + /// + /// Set by the emulator image and never by a production one — and because + /// the image's environment is measured, PCR0 itself tells a client which + /// kind it is talking to. + UnsignedEmulator, +} + +impl ReceiptTrust { + pub fn parse(s: &str) -> Result { + match s.trim().to_ascii_lowercase().as_str() { + "required" | "signed" => Ok(ReceiptTrust::Required), + "unsigned-emulator" | "unsigned" => Ok(ReceiptTrust::UnsignedEmulator), + other => Err(format!( + "expected one of required, unsigned-emulator; got {other:?}" + )), + } + } +} + +/// Everything the boot machine needs that is not already in [`MountConfig`]. +/// +/// There is deliberately no "adopt this existing filesystem" escape hatch. One +/// would authorise whatever state happened to be present — the single thing +/// this module exists to refuse — and it could only ever be justified by a +/// deployment that predates receipts. None does: receipts were here before the +/// first enclave was. Anything that needs adopting is, by construction, state +/// of unknown origin. +pub struct BootConfig { + pub trust: ReceiptTrust, +} + +/// Read an origin record, distinguishing "absent" from "could not read" — and +/// both of those from "hidden". +/// +/// [`Backend::get_retained_blob`] rather than `get_blob`, because this is the +/// one place in the system where **absence is a decision**. Object Lock makes +/// these records indestructible but not unhideable: a `DeleteObject` without a +/// version id writes a delete marker, an ordinary `GetObject` then answers +/// `NoSuchKey`, and a conditional PUT over the marker succeeds because the +/// current version is no longer an object. Hide the receipt and the sealed key +/// together and this function would report `(None, None)` — genesis — and the +/// enclave would create a second filesystem beside the one it was hiding, +/// with every signature along the way valid. +/// +/// Every other record here is content-verified, so hiding is the only attack +/// that has no signature to fail. Reading the retained version is what makes it +/// fail instead. +async fn maybe_get(backend: &Arc, key: &str) -> Result>> { + match backend.get_retained_blob(key).await { + Ok(out) => Ok(Some(out.body.to_vec())), + Err(FsError::NotFound) => Ok(None), + Err(e) => Err(anyhow::Error::msg(e.to_string())).with_context(|| format!("reading {key}")), + } +} + +/// Verify a receipt and return what it committed to. +fn open_receipt( + document: &[u8], + trust: ReceiptTrust, + purpose: &str, + expected: &Expectations, +) -> Result { + let verified = match trust { + ReceiptTrust::Required => verify_as_signed(document, AWS_NITRO_ROOT_G1_PEM.as_bytes())?, + // The emulator's NSM does not sign, so there is nothing to verify and + // the contents are read as-is. Only an image built for the emulator + // reaches here, and because the image's environment is measured, PCR0 + // says which kind of image a client is talking to. + ReceiptTrust::UnsignedEmulator => nitro_attestation::Verified { + document: nitro_attestation::parse(document).context("parsing the receipt")?, + trust: Trust::Unsigned, + }, + }; + // `max_age` is deliberately unset: a state-origin receipt is a statement + // about an origin, not a proof of liveness. It is *supposed* to be old, + // and expiring one would make a filesystem unmountable by the passage of + // time. + verified + .expect(expected, std::time::SystemTime::now()) + .with_context(|| format!("the {purpose} receipt does not say what it must"))?; + Ok(verified) +} + +/// Verify a stored document's signature and chain **as of when it was signed**. +/// +/// Not as of now. The certificates in an attestation document's chain are +/// short-lived — far shorter than a filesystem's life — so a receipt checked +/// against the current time stops verifying soon after it is written, and the +/// filesystem it guards becomes unmountable by the passage of time: the failure +/// `max_age` is left unset to avoid, arriving by another route. +/// +/// The timestamp is part of the signed payload. It is read before the signature +/// is checked only to choose the moment to check at; `verify` then fails unless +/// the chain was valid at that moment *and* the signature covers that +/// timestamp. +fn verify_as_signed(document: &[u8], trust_root: &[u8]) -> Result { + let signed_at = nitro_attestation::parse(document) + .context("parsing the receipt")? + .timestamp(); + let verified = nitro_attestation::verify( + document, + &VerifyOptions { + trust_root: trust_root.to_vec(), + now: signed_at, + allow_untrusted_root: false, + }, + ) + .context("verifying the receipt")?; + if verified.trust != Trust::ChainVerified { + bail!( + "receipt is {:?} rather than chain-verified; this image requires a receipt \ + signed by AWS", + verified.trust + ); + } + Ok(verified) +} + +/// Decide the mode, resolve the key, and mount or create. +pub async fn boot( + backends: &Backends, + mount_config: &MountConfig, + boot_config: &BootConfig, + nsm: &Arc, + key_source: &dyn MasterKeySource, +) -> Result { + let prefix = &mount_config.bucket_prefix; + let fs_uuid = &mount_config.fs_id; + let roots = &backends.roots; + + // Before anything is read from the store or asked of KMS. + let pair = Pair::read(nsm.as_ref())?; + + let receipt = maybe_get(roots, &receipt_key(prefix, fs_uuid)).await?; + let sealed = maybe_get(roots, &sealed_key_key(prefix, fs_uuid)) + .await? + .map(SealedKey::from_bytes); + + match (receipt, sealed) { + // ---- resume or upgrade --------------------------------------------- + (Some(receipt), Some(sealed)) => { + // Under `kms` this is where the key policy decides which runtime + // and guest may hold this state, and the only place: a pair it does + // not name gets no key and goes no further. + let master = key_source + .open(&sealed) + .await + .context("opening this filesystem's master key")?; + let mounted = crate::mount::mount_existing(backends, mount_config, &master).await?; + + let identity = identity_of(&mounted, mount_config, &sealed).await?; + let state_root = identity.state_root(); + + verify_origin(&receipt, boot_config.trust, &state_root)?; + let first = record_pair( + backends, + mount_config, + boot_config.trust, + nsm.as_ref(), + &pair, + &state_root, + ) + .await?; + + Ok(Booted { + mode: if first { + BootMode::Upgrade + } else { + BootMode::Resume + }, + mounted, + state_root, + pair, + master, + }) + } + + // ---- genesis ------------------------------------------------------- + (None, None) => { + genesis( + backends, + mount_config, + boot_config.trust, + nsm, + key_source, + pair, + ) + .await + } + + // ---- the attack ---------------------------------------------------- + (Some(_), None) => bail!( + "this filesystem has a state-origin receipt but no sealed key. \ + Either the key was deleted — which Object Lock should have \ + prevented — or the store is not the one the receipt describes. \ + Refusing to boot rather than creating a second filesystem." + ), + (None, Some(_)) => bail!( + "this filesystem has a sealed key but no state-origin receipt. \ + Genesis writes the receipt last, so this is either a genesis \ + interrupted between the two, or a store with its receipt hidden. \ + Nothing here can tell those apart, and treating it as the first \ + would serve state no enclave ever accounted for. Refusing to boot." + ), + } +} + +/// Recompute the identity of a filesystem that is already mounted. +async fn identity_of( + mounted: &Mounted, + config: &MountConfig, + sealed: &SealedKey, +) -> Result { + // The seq-0 record, read past the session floor because it is history + // rather than the live tip. + let genesis_root = mounted + .fs + .store() + .snapshot_root(0) + .await + .map_err(|e| anyhow::anyhow!("reading the genesis root record: {e}"))?; + + Ok(StateIdentity { + fs_uuid: config.fs_id, + data_bucket: config.bucket.clone(), + roots_bucket: config + .roots_bucket + .clone() + .unwrap_or_else(|| config.bucket.clone()), + bucket_prefix: config.bucket_prefix.clone(), + genesis_root_hash: *genesis_root.hash().as_bytes(), + sealed_key_sha256: sealed.sha256(), + }) +} + +/// Check the origin receipt names the state that was just loaded. +/// +/// Not who wrote it. Any runtime and guest may have run genesis, because by +/// this point KMS has released the key to *this* pair, and that is the decision +/// that matters — see the module docs. What KMS cannot say is whether the state +/// loaded is the state genesis recorded, and that is the question here. +fn verify_origin(receipt: &[u8], trust: ReceiptTrust, state_root: &[u8; 32]) -> Result<()> { + open_receipt( + receipt, + trust, + "state-origin", + &Expectations::default().user_data(receipt_payload(PURPOSE_STATE_ORIGIN, state_root)), + )?; + Ok(()) +} + +/// Check a pair record says this runtime and guest held this state. +fn verify_pair_record( + document: &[u8], + trust: ReceiptTrust, + pair: &Pair, + state_root: &[u8; 32], +) -> Result<()> { + open_receipt( + document, + trust, + "pair", + &Expectations::default() + .pcr0(pair.pcr0.to_vec()) + .pcr(PCR_GUEST as u32, pair.pcr16.to_vec()) + .user_data(receipt_payload(PURPOSE_PAIR, state_root)), + )?; + Ok(()) +} + +/// Make sure this pair has a record against this state, returning whether this +/// boot is the pair's first. +/// +/// A record that is already there is **verified, not merely found**. Its key +/// is derivable by anyone who can write to the roots bucket — the parent +/// included — so an object planted there ahead of a pair's first boot would +/// otherwise stand in for the real record for good, Object Lock keeping it as +/// faithfully as it would the genuine one. +async fn record_pair( + backends: &Backends, + config: &MountConfig, + trust: ReceiptTrust, + nsm: &dyn Nsm, + pair: &Pair, + state_root: &[u8; 32], +) -> Result { + let key = pair.record_key(&config.bucket_prefix, &config.fs_id); + + if let Some(existing) = maybe_get(&backends.roots, &key).await? { + verify_pair_record(&existing, trust, pair, state_root).with_context(|| { + format!( + "the record at {key} does not describe this runtime and guest holding this \ + state. Refusing to boot rather than leave it standing as this pair's record." + ) + })?; + return Ok(false); + } + + let document = nsm + .attest(&AttestationRequest::with_user_data(receipt_payload( + PURPOSE_PAIR, + state_root, + ))) + .context("asking the NSM for a pair record")?; + // Before it is stored. A document that did not carry both registers would + // be a record of nothing, locked in place for the retention period — and on + // real hardware, this is where a guest register missing from documents + // would first show. + verify_pair_record(&document, trust, pair, state_root) + .context("the NSM's own document does not carry this runtime and guest")?; + + match backends + .roots + .put_blob_if_not_exists(locked(&key, document, config)) + .await + { + Ok(_) => {} + // Another first boot of this pair got there first. Its record stands, + // provided it says what this one would have. + Err(FsError::AlreadyExists) => { + let theirs = maybe_get(&backends.roots, &key) + .await? + .with_context(|| format!("{key} was reported present and then was not"))?; + verify_pair_record(&theirs, trust, pair, state_root).with_context(|| { + format!( + "another writer's record at {key} does not describe this runtime and \ + guest holding this state. Refusing to boot." + ) + })?; + } + Err(other) => bail!("writing the pair record {key}: {other}"), + } + + tracing::info!( + pcr0 = %hex::encode(pair.pcr0), + pcr16 = %hex::encode(pair.pcr16), + record = %key, + "first boot of this runtime and guest on this state; recorded" + ); + Ok(true) +} + +/// Create a filesystem, the receipt that authorises every later boot, and the +/// first pair record. +/// +/// The ordering is load-bearing and only the last steps are atomic: mint, seal, +/// persist the key, create the filesystem, then write the receipt with a +/// conditional PUT. A crash before the receipt leaves a filesystem with no +/// receipt, which the next boot refuses — deliberately, because that state is +/// indistinguishable from a store whose receipt has been hidden. +/// +/// The pair record comes after the receipt, so the receipt stays the last thing +/// whose absence means "unfinished". A genesis interrupted between the two +/// leaves a store the next boot resumes; that boot writes the record and +/// reports an upgrade, which for once is not one. +async fn genesis( + backends: &Backends, + config: &MountConfig, + trust: ReceiptTrust, + nsm: &Arc, + key_source: &dyn MasterKeySource, + pair: Pair, +) -> Result { + tracing::info!( + pcr0 = %hex::encode(pair.pcr0), + pcr16 = %hex::encode(pair.pcr16), + "no filesystem here: creating one and recording its origin" + ); + + let (master, sealed) = key_source.mint().await.context("minting a master key")?; + + // Written first and *conditionally*: this is the genesis lease. Two cold + // boots race here and exactly one wins, so they cannot create divergent + // filesystems under different keys. + let key_object = sealed_key_key(&config.bucket_prefix, &config.fs_id); + backends + .roots + .put_blob_if_not_exists(locked(&key_object, sealed.as_bytes().to_vec(), config)) + .await + .map_err(|e| match e { + // `AlreadyExists`, not `Conflict`: that is what both backends + // return from a failed conditional PUT (`backend/mod.rs`), and + // matching `Conflict` here meant this arm never fired — a genesis + // race reported the generic message instead of the specific one. + FsError::AlreadyExists => anyhow::anyhow!( + "another enclave is performing genesis on this filesystem right now" + ), + other => anyhow::anyhow!("storing the sealed master key: {other}"), + })?; + + let mounted = crate::mount::create(backends, config, &master).await?; + let identity = identity_of(&mounted, config, &sealed).await?; + let state_root = identity.state_root(); + + let document = nsm + .attest(&AttestationRequest::with_user_data(receipt_payload( + PURPOSE_STATE_ORIGIN, + &state_root, + ))) + .context("asking the NSM for a state-origin receipt")?; + + let receipt_object = receipt_key(&config.bucket_prefix, &config.fs_id); + backends + .roots + .put_blob_if_not_exists(locked(&receipt_object, document, config)) + .await + .map_err(|e| anyhow::anyhow!("writing the state-origin receipt: {e}"))?; + + record_pair(backends, config, trust, nsm.as_ref(), &pair, &state_root).await?; + + tracing::info!( + state_root = %hex::encode(state_root), + "genesis complete; this filesystem now has an attested origin" + ); + + Ok(Booted { + mode: BootMode::Genesis, + mounted, + state_root, + pair, + master, + }) +} + +/// Everything the boot machine writes goes under the same retention as the +/// anchor chain: undeletable is the whole point. +fn locked(key: &str, body: Vec, _config: &MountConfig) -> PutBlobInput { + let mut input = PutBlobInput::new(key, body.into()); + if let Some(retention) = s3fs_core::store::StoreConfig::default().root_retention { + input = input.with_object_lock(ObjectLock { + mode: s3fs_core::backend::ObjectLockMode::Compliance, + retain_until: std::time::SystemTime::now() + retention, + }); + } + input +} + +#[cfg(test)] +mod tests { + use super::*; + use nitro_nsm::fake::FakeNsm; + + fn identity() -> StateIdentity { + StateIdentity { + fs_uuid: [1u8; 16], + data_bucket: "data".into(), + roots_bucket: "roots".into(), + bucket_prefix: String::new(), + genesis_root_hash: [2u8; 32], + sealed_key_sha256: [3u8; 32], + } + } + + fn pair(pcr0: u8, pcr16: u8) -> Pair { + Pair { + pcr0: [pcr0; 48], + pcr16: [pcr16; 48], + } + } + + #[test] + fn the_state_root_is_deterministic() { + assert_eq!(identity().state_root(), identity().state_root()); + } + + /// Every field is something an attacker could otherwise vary while leaving + /// the rest of the design intact, so every field must change the hash. + #[test] + fn every_field_changes_the_state_root() { + let base = identity().state_root(); + + let mut a = identity(); + a.fs_uuid = [9u8; 16]; + assert_ne!(a.state_root(), base, "filesystem id"); + + let mut b = identity(); + b.data_bucket = "elsewhere".into(); + assert_ne!(b.state_root(), base, "data bucket"); + + let mut c = identity(); + c.roots_bucket = "elsewhere".into(); + assert_ne!(c.state_root(), base, "roots bucket"); + + let mut d = identity(); + d.bucket_prefix = "other/".into(); + assert_ne!(d.state_root(), base, "prefix"); + + let mut e = identity(); + e.genesis_root_hash = [9u8; 32]; + assert_ne!(e.state_root(), base, "history"); + + let mut f = identity(); + f.sealed_key_sha256 = [9u8; 32]; + assert_ne!(f.state_root(), base, "key"); + } + + /// Swapping two string fields must not produce the same pre-image. CBOR + /// length-prefixes each, so it does not — worth pinning, because a + /// concatenation-based encoding would. + #[test] + fn swapping_fields_is_not_the_same_state() { + let mut swapped = identity(); + std::mem::swap(&mut swapped.data_bucket, &mut swapped.roots_bucket); + assert_ne!(swapped.state_root(), identity().state_root()); + } + + #[test] + fn the_payload_distinguishes_its_purposes() { + let root = [4u8; 32]; + assert_ne!( + receipt_payload(PURPOSE_STATE_ORIGIN, &root), + receipt_payload(PURPOSE_PAIR, &root), + "a pair record must not verify as a state-origin receipt" + ); + } + + /// Byte for byte, what state-origin receipts have always carried. Every + /// receipt already written commits to this, so it cannot move without + /// making every existing filesystem unmountable. + #[test] + fn the_state_origin_payload_has_not_changed() { + let root = [4u8; 32]; + let mut expected = vec![0xa2]; + expected.push(0x67); + expected.extend_from_slice(b"purpose"); + expected.push(0x71); + expected.extend_from_slice(b"s3fs-state-origin"); + expected.push(0x6a); + expected.extend_from_slice(b"state_root"); + expected.extend_from_slice(&[0x58, 0x20]); + expected.extend_from_slice(&root); + assert_eq!(receipt_payload(PURPOSE_STATE_ORIGIN, &root), expected); + } + + /// Keys must be derivable from what the enclave already knows — that is + /// what makes a 404 an answer rather than a failure to look in the right + /// place. + #[test] + fn object_keys_are_derived_and_distinct() { + let uuid = [7u8; 16]; + let r = receipt_key("p/", &uuid); + let k = sealed_key_key("p/", &uuid); + let p = pair(1, 2).record_key("p/", &uuid); + assert_ne!(r, k); + assert_ne!(r, p); + assert_ne!(k, p); + assert!(r.starts_with("p/")); + assert!(p.starts_with("p/origin/")); + assert_eq!(r, receipt_key("p/", &uuid), "must be deterministic"); + assert_eq!( + p, + pair(1, 2).record_key("p/", &uuid), + "must be deterministic" + ); + } + + /// One record per pair, so the key has to change with either register — + /// and with which register holds which value. + #[test] + fn a_pair_record_key_names_both_measurements() { + let uuid = [7u8; 16]; + let base = pair(1, 2).record_key("", &uuid); + assert_ne!(pair(9, 2).record_key("", &uuid), base, "another runtime"); + assert_ne!(pair(1, 9).record_key("", &uuid), base, "another guest"); + assert_ne!(pair(2, 1).record_key("", &uuid), base, "registers swapped"); + assert_ne!( + pair(1, 2).record_key("", &[8u8; 16]), + base, + "another filesystem" + ); + } + + #[test] + fn a_measured_guest_is_read_back_as_the_pair() { + let nsm = FakeNsm::new(); + let pcr16 = crate::guest::measure_guest(&nsm, b"a guest").unwrap(); + let read = Pair::read(&nsm).unwrap(); + assert_eq!(read.pcr16, pcr16); + assert_eq!(read.pcr0, [0x10; 48], "the fake's PCR0"); + } + + /// Boot's first refusal, and the one that keeps an unmeasured guest from + /// ever reaching KMS. + #[test] + fn an_unmeasured_guest_is_not_a_pair() { + let err = Pair::read(&FakeNsm::new()).unwrap_err(); + assert!( + format!("{err:#}").contains("PCR16 is not locked"), + "{err:#}" + ); + } + + #[test] + fn a_locked_but_empty_guest_register_is_not_a_pair() { + let nsm = FakeNsm::new(); + nsm.lock_pcr(PCR_GUEST).unwrap(); + let err = Pair::read(&nsm).unwrap_err(); + assert!(format!("{err:#}").contains("never extended"), "{err:#}"); + } + + /// A signed document stamped at `signed_at`, from a chain valid only for + /// the hour after `valid_from`. + fn stamped( + valid_from: std::time::SystemTime, + signed_at: std::time::SystemTime, + ) -> (nitro_attestation::testing::TestChain, Vec) { + use std::time::{Duration, UNIX_EPOCH}; + let chain = nitro_attestation::testing::TestChain::with_validity( + valid_from, + valid_from + Duration::from_secs(3600), + ) + .unwrap(); + let mut document = nitro_attestation::parse( + &chain + .document(Some(b"payload".to_vec()), None, [0xab; 48]) + .unwrap(), + ) + .unwrap(); + document.timestamp_ms = signed_at.duration_since(UNIX_EPOCH).unwrap().as_millis() as u64; + let receipt = chain.document_from(document).unwrap(); + (chain, receipt) + } + + /// A receipt outlives the certificates that signed it. Judged against the + /// current time it would stop verifying soon after it was written, and the + /// filesystem it guards with it. + #[test] + fn a_receipt_is_judged_as_of_when_it_was_signed() { + use std::time::{Duration, SystemTime}; + let long_ago = SystemTime::now() - Duration::from_secs(3600 * 24 * 90); + let (chain, receipt) = stamped(long_ago, long_ago + Duration::from_secs(60)); + + assert!( + nitro_attestation::verify( + &receipt, + &VerifyOptions { + trust_root: chain.root_der().to_vec(), + now: SystemTime::now(), + allow_untrusted_root: false, + }, + ) + .is_err(), + "the chain has expired, so a check against now must refuse it" + ); + let verified = + verify_as_signed(&receipt, chain.root_der()).expect("valid when it was signed"); + assert_eq!( + verified.document.user_data.as_deref(), + Some(&b"payload"[..]) + ); + } + + /// And the moment a receipt claims has to be one its chain was valid at. A + /// document stamped outside that window is refused, whatever the time now. + #[test] + fn a_receipt_stamped_outside_its_chain_is_refused() { + use std::time::{Duration, SystemTime}; + let long_ago = SystemTime::now() - Duration::from_secs(3600 * 24 * 90); + let (chain, receipt) = stamped(long_ago, SystemTime::now()); + assert!(verify_as_signed(&receipt, chain.root_der()).is_err()); + } + + #[test] + fn trust_parses_the_documented_values() { + assert_eq!(ReceiptTrust::parse("required"), Ok(ReceiptTrust::Required)); + assert_eq!( + ReceiptTrust::parse("unsigned-emulator"), + Ok(ReceiptTrust::UnsignedEmulator) + ); + assert!(ReceiptTrust::parse("none").is_err()); + } +} diff --git a/runtime/src/clock.rs b/runtime/src/clock.rs new file mode 100644 index 0000000..a5c7269 --- /dev/null +++ b/runtime/src/clock.rs @@ -0,0 +1,359 @@ +//! Where the guest's idea of "now" comes from. +//! +//! An enclave's system clock is not its own. It is seeded by the hypervisor at +//! boot, has no NTP, and drifts — and the party that sets it is the parent +//! instance, which is exactly the party the enclave exists to distrust. AWS +//! addresses this by exposing the Nitro card's PTP hardware clock, synchronised +//! to the Amazon Time Sync Service, at `/dev/ptp0`. +//! +//! Reading it uses the POSIX *dynamic clock* mechanism: open the character +//! device, then derive a clock id from the file descriptor and call +//! `clock_gettime` on that. The encoding is `((~fd) << 3) | 3`, which +//! [`rustix::time::clock_gettime_dynamic`] performs for us — worth using rather +//! than open-coding, since getting the bit twiddle wrong yields a valid-looking +//! clock id for some *other* clock rather than an error. + +use std::fmt; +use std::fs::File; +use std::os::fd::{AsFd, OwnedFd}; +use std::path::{Path, PathBuf}; +use std::sync::Mutex; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use anyhow::{Context, Result}; +use rustix::time::{clock_gettime_dynamic, DynamicClockId}; + +/// The conventional device for the Nitro PTP hardware clock. +pub const DEFAULT_PTP_DEVICE: &str = "/dev/ptp0"; + +/// A source of wall-clock time. +pub trait TrustedClock: Send + Sync + fmt::Debug { + /// Time since the Unix epoch. + fn now(&self) -> Result; + + /// Granularity of this clock. + fn resolution(&self) -> Duration; + + /// Short description for the startup log. A deployment reading its own + /// logs must be able to tell which clock it actually got. + fn describe(&self) -> String; +} + +/// The PTP hardware clock, read through its character device. +pub struct PtpClock { + device: PathBuf, + fd: OwnedFd, +} + +impl PtpClock { + /// Open the device and take one reading, so a device that exists but + /// cannot be read fails here rather than on the guest's first call. + pub fn open(device: impl AsRef) -> Result { + let device = device.as_ref().to_path_buf(); + let file = File::open(&device) + .with_context(|| format!("opening PTP clock device {}", device.display()))?; + let clock = PtpClock { + device, + fd: OwnedFd::from(file), + }; + clock + .now() + .with_context(|| format!("reading PTP clock device {}", clock.device.display()))?; + Ok(clock) + } + + pub fn device(&self) -> &Path { + &self.device + } +} + +impl fmt::Debug for PtpClock { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("PtpClock") + .field("device", &self.device) + .finish() + } +} + +impl TrustedClock for PtpClock { + fn now(&self) -> Result { + let ts = clock_gettime_dynamic(DynamicClockId::Dynamic(self.fd.as_fd())) + .context("clock_gettime on the PTP dynamic clock id")?; + // A PHC counts from the Unix epoch like CLOCK_REALTIME, so a negative + // reading means the device is not what we think it is. + let secs = u64::try_from(ts.tv_sec) + .map_err(|_| anyhow::anyhow!("PTP clock returned a pre-epoch time"))?; + Ok(Duration::new(secs, ts.tv_nsec as u32)) + } + + fn resolution(&self) -> Duration { + // A PHC is nanosecond-granular in its interface. The underlying + // oscillator is coarser, but nothing here can measure that. + Duration::from_nanos(1) + } + + fn describe(&self) -> String { + format!("PTP hardware clock ({})", self.device.display()) + } +} + +/// The host's `CLOCK_REALTIME`. +/// +/// Fine for development. Inside an enclave it is whatever the hypervisor last +/// set, which is why using it there is worth a warning. +#[derive(Debug, Default)] +pub struct HostClock; + +impl TrustedClock for HostClock { + fn now(&self) -> Result { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .context("host clock is before the Unix epoch") + } + + fn resolution(&self) -> Duration { + Duration::from_nanos(1) + } + + fn describe(&self) -> String { + "host system clock (untrusted)".to_string() + } +} + +/// Which clock to use. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum ClockSource { + /// PTP if the device opens, host clock otherwise. + #[default] + Auto, + /// PTP, or refuse to start. + Ptp, + /// Host clock, unconditionally. + Host, +} + +impl ClockSource { + pub fn parse(s: &str) -> Result { + match s.trim().to_ascii_lowercase().as_str() { + "auto" => Ok(ClockSource::Auto), + "ptp" => Ok(ClockSource::Ptp), + "host" => Ok(ClockSource::Host), + other => Err(format!("expected one of auto, ptp, host; got {other:?}")), + } + } +} + +/// Resolve a clock source into a clock, logging which one was chosen. +/// +/// `Auto` falls back at `warn` rather than silently: a misconfigured enclave +/// running on host time must not look identical in the logs to a correctly +/// configured one. +pub fn open_clock(source: ClockSource, device: &Path) -> Result> { + match source { + ClockSource::Host => { + tracing::warn!(clock = "host", "using the host clock; time is untrusted"); + Ok(Box::new(HostClock)) + } + ClockSource::Ptp => { + let clock = PtpClock::open(device) + .context("clock source 'ptp' was required but the device could not be opened")?; + tracing::info!(clock = %clock.describe(), "clock source"); + Ok(Box::new(clock)) + } + ClockSource::Auto => match PtpClock::open(device) { + Ok(clock) => { + tracing::info!(clock = %clock.describe(), "clock source"); + Ok(Box::new(clock)) + } + Err(e) => { + tracing::warn!( + device = %device.display(), + error = format!("{e:#}"), + "PTP clock unavailable, falling back to the host clock; time is untrusted" + ); + Ok(Box::new(HostClock)) + } + }, + } +} + +/// Adapts a [`TrustedClock`] to what `wasmtime-wasi` wants for +/// `wasi:clocks/wall-clock`. +/// +/// `HostWallClock::now` cannot fail, so this has to decide what a failed device +/// read means mid-run. It returns the last good reading and logs once. A guest +/// should not trap because a driver hiccuped, and a zero timestamp would be far +/// more damaging downstream — expired certificates, rejected tokens, dates in +/// 1970 — than one that is a few milliseconds stale. +pub struct WallClockAdapter { + clock: Box, + last_good: Mutex, +} + +impl WallClockAdapter { + pub fn new(clock: Box) -> Result { + let initial = clock.now()?; + Ok(WallClockAdapter { + clock, + last_good: Mutex::new(initial), + }) + } +} + +impl fmt::Debug for WallClockAdapter { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("WallClockAdapter") + .field("clock", &self.clock) + .finish() + } +} + +impl wasmtime_wasi::HostWallClock for WallClockAdapter { + fn resolution(&self) -> Duration { + self.clock.resolution() + } + + fn now(&self) -> Duration { + match self.clock.now() { + Ok(now) => { + *self.last_good.lock().expect("clock mutex poisoned") = now; + now + } + Err(e) => { + let stale = *self.last_good.lock().expect("clock mutex poisoned"); + tracing::warn!( + error = format!("{e:#}"), + "clock read failed; serving the last good reading" + ); + stale + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; + use std::sync::Arc; + use wasmtime_wasi::HostWallClock; + + /// A clock that can be made to fail on demand. + #[derive(Debug)] + struct FakeClock { + nanos: AtomicU64, + fail: AtomicBool, + } + + impl FakeClock { + fn new(nanos: u64) -> Arc { + Arc::new(FakeClock { + nanos: AtomicU64::new(nanos), + fail: AtomicBool::new(false), + }) + } + } + + // Implemented on the handle so a test can keep one and still hand the + // adapter ownership. + impl TrustedClock for Arc { + fn now(&self) -> Result { + if self.fail.load(Ordering::SeqCst) { + anyhow::bail!("device read failed"); + } + Ok(Duration::from_nanos(self.nanos.load(Ordering::SeqCst))) + } + fn resolution(&self) -> Duration { + Duration::from_nanos(7) + } + fn describe(&self) -> String { + "fake".into() + } + } + + #[test] + fn clock_source_parses_the_documented_values() { + assert_eq!(ClockSource::parse("auto"), Ok(ClockSource::Auto)); + assert_eq!(ClockSource::parse("PTP"), Ok(ClockSource::Ptp)); + assert_eq!(ClockSource::parse(" host "), Ok(ClockSource::Host)); + assert_eq!(ClockSource::default(), ClockSource::Auto); + + let err = ClockSource::parse("gps").unwrap_err(); + assert!(err.contains("gps"), "{err}"); + assert!(err.contains("auto"), "{err}"); + } + + #[test] + fn the_adapter_passes_readings_and_resolution_through() { + let fake = FakeClock::new(1_700_000_000_000_000_000); + let adapter = WallClockAdapter::new(Box::new(fake)).unwrap(); + assert_eq!( + adapter.now(), + Duration::from_nanos(1_700_000_000_000_000_000) + ); + assert_eq!(adapter.resolution(), Duration::from_nanos(7)); + } + + /// The property that matters when a device read fails mid-run: a guest gets + /// a slightly stale answer, never a zero one, and never a trap. + #[test] + fn a_failed_read_serves_the_last_good_value() { + let fake = FakeClock::new(1_000); + let adapter = WallClockAdapter::new(Box::new(fake.clone())).unwrap(); + + // Advance, read, then break the device. + fake.nanos.store(5_000, Ordering::SeqCst); + assert_eq!(adapter.now(), Duration::from_nanos(5_000)); + + fake.fail.store(true, Ordering::SeqCst); + assert_eq!( + adapter.now(), + Duration::from_nanos(5_000), + "must serve the last good reading, not zero" + ); + assert_ne!(adapter.now(), Duration::ZERO); + + // Recovery restores live readings. + fake.fail.store(false, Ordering::SeqCst); + fake.nanos.store(9_000, Ordering::SeqCst); + assert_eq!(adapter.now(), Duration::from_nanos(9_000)); + } + + #[test] + fn a_clock_that_cannot_be_read_at_all_fails_to_construct() { + let fake = FakeClock::new(0); + fake.fail.store(true, Ordering::SeqCst); + assert!(WallClockAdapter::new(Box::new(fake.clone())).is_err()); + } + + #[test] + fn opening_a_missing_device_names_it() { + let err = PtpClock::open("/dev/definitely-not-a-ptp-device").unwrap_err(); + let rendered = format!("{err:#}"); + assert!( + rendered.contains("/dev/definitely-not-a-ptp-device"), + "{rendered}" + ); + } + + /// `auto` must degrade rather than refuse; `ptp` must refuse rather than + /// degrade. That asymmetry is the whole point of having both. + #[test] + fn auto_falls_back_but_ptp_does_not() { + let missing = Path::new("/dev/definitely-not-a-ptp-device"); + let fallback = open_clock(ClockSource::Auto, missing).expect("auto must fall back"); + assert!(fallback.describe().contains("host")); + + assert!(open_clock(ClockSource::Ptp, missing).is_err()); + } + + #[test] + fn the_host_clock_is_sane_and_labelled_untrusted() { + let clock = HostClock; + let now = clock.now().unwrap(); + // Somewhere after 2020 and before 2100, i.e. a real wall clock. + assert!(now.as_secs() > 1_577_836_800, "{now:?}"); + assert!(now.as_secs() < 4_102_444_800, "{now:?}"); + assert!(clock.describe().contains("untrusted")); + } +} diff --git a/runtime/src/env.rs b/runtime/src/env.rs new file mode 100644 index 0000000..68e25cd --- /dev/null +++ b/runtime/src/env.rs @@ -0,0 +1,285 @@ +//! What the guest's environment contains, and what it must never contain. +//! +//! This process holds the filesystem master key, and under the attestation +//! milestone it will hold KMS-released credentials. The guest is the code the +//! enclave exists to contain. So the environment it sees is the host's, minus +//! two categories: +//! +//! - **Credentials**, obviously. +//! - **The runtime's own configuration**, which tells the guest where its data +//! lives and is of no use to it — an enclave guest has no network of its own. +//! +//! Both fall under one rule that is easy to state and hard to get wrong: +//! nothing whose name begins with `AWS_` or `S3FS_` is inherited. That covers +//! `S3FS_MASTER_KEY` and the whole AWS credential set without anyone having to +//! enumerate them, and it keeps working when a new one is added. +//! +//! An operator who genuinely needs one of those names can still set it +//! explicitly. Explicit is a decision; inheritance is an accident waiting to +//! happen. + +use std::collections::BTreeMap; + +/// Name prefixes that are never inherited from the host. +pub const DENIED_PREFIXES: &[&str] = &["AWS_", "S3FS_"]; + +/// Names that are credentials wherever they appear. Setting one explicitly is +/// permitted — the operator asked for it — but it is worth saying out loud. +const CREDENTIAL_NAMES: &[&str] = &[ + "AWS_ACCESS_KEY_ID", + "AWS_SECRET_ACCESS_KEY", + "AWS_SESSION_TOKEN", + "AWS_SECURITY_TOKEN", + "S3FS_MASTER_KEY", +]; + +fn is_denied(name: &str) -> bool { + DENIED_PREFIXES.iter().any(|p| name.starts_with(p)) +} + +fn is_credential(name: &str) -> bool { + CREDENTIAL_NAMES.contains(&name) +} + +/// How the guest's environment is assembled. +#[derive(Debug, Clone)] +pub struct GuestEnvPolicy { + /// Inherit the host environment (minus [`DENIED_PREFIXES`]). + pub inherit: bool, + /// Explicit entries, as `NAME` (inherit that one by name) or `NAME=VALUE`. + /// Applied after inheritance, so they override it. + pub explicit: Vec, +} + +impl Default for GuestEnvPolicy { + fn default() -> Self { + GuestEnvPolicy { + inherit: true, + explicit: Vec::new(), + } + } +} + +impl GuestEnvPolicy { + /// Inherit nothing; the guest sees only what is named explicitly. + pub fn explicit_only(explicit: Vec) -> Self { + GuestEnvPolicy { + inherit: false, + explicit, + } + } + + /// Build the environment, reading the host's via `std::env::vars`. + pub fn build(&self) -> anyhow::Result> { + self.build_from(std::env::vars().collect::>().as_slice()) + } + + /// Build against an explicit host environment. Separated so the policy is + /// testable without mutating the process, which no test should have to do. + pub fn build_from(&self, host: &[(String, String)]) -> anyhow::Result> { + // Sorted so the guest's environment is deterministic regardless of the + // host's iteration order. + let mut out: BTreeMap = BTreeMap::new(); + let lookup = |name: &str| -> Option<&str> { + host.iter() + .find(|(k, _)| k == name) + .map(|(_, v)| v.as_str()) + }; + + if self.inherit { + let mut denied: Vec<&str> = Vec::new(); + for (name, value) in host { + if is_denied(name) { + denied.push(name); + continue; + } + out.insert(name.clone(), value.clone()); + } + if !denied.is_empty() { + // Say what was withheld. A guest missing a variable should be + // diagnosable from the log rather than mysterious. + denied.sort_unstable(); + tracing::info!( + withheld = denied.join(", "), + "guest environment: withheld runtime configuration and credentials" + ); + } + } + + for spec in &self.explicit { + let (name, value) = match spec.split_once('=') { + Some((name, value)) => (name, Some(value.to_string())), + None => (spec.as_str(), None), + }; + if name.is_empty() { + anyhow::bail!("guest-env: empty variable name in {spec:?}"); + } + let value = match value { + Some(v) => v, + None => match lookup(name) { + Some(v) => v.to_string(), + // A bare name that is unset on the host is skipped rather + // than passed as empty, so the guest can tell "unset" from + // "set to nothing". + None => { + tracing::debug!(name, "guest-env: not set on host, skipping"); + continue; + } + }, + }; + if is_credential(name) { + tracing::warn!( + name, + "guest-env: exposing a credential to guest code, because it was named explicitly" + ); + } + out.insert(name.to_string(), value); + } + + Ok(out.into_iter().collect()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn host() -> Vec<(String, String)> { + [ + ("LANG", "en_GB.UTF-8"), + ("RUST_LOG", "info"), + ("MY_APP_ENDPOINT", "https://example.invalid"), + ("AWS_SECRET_ACCESS_KEY", "super-secret"), + ("AWS_ACCESS_KEY_ID", "AKIA..."), + ("AWS_SESSION_TOKEN", "token"), + ("AWS_REGION", "eu-west-2"), + ("S3FS_MASTER_KEY", "0011..."), + ("S3FS_BUCKET", "prod-data"), + ] + .into_iter() + .map(|(k, v)| (k.to_string(), v.to_string())) + .collect() + } + + fn names(env: &[(String, String)]) -> Vec<&str> { + env.iter().map(|(k, _)| k.as_str()).collect() + } + + #[test] + fn inheritance_passes_ordinary_variables() { + let env = GuestEnvPolicy::default().build_from(&host()).unwrap(); + assert_eq!(names(&env), vec!["LANG", "MY_APP_ENDPOINT", "RUST_LOG"]); + } + + /// The property the whole module exists for. + #[test] + fn no_credential_or_runtime_variable_is_ever_inherited() { + let env = GuestEnvPolicy::default().build_from(&host()).unwrap(); + for (name, _) in &env { + assert!( + !name.starts_with("AWS_") && !name.starts_with("S3FS_"), + "{name} reached the guest by inheritance" + ); + } + assert!(!env.iter().any(|(_, v)| v == "super-secret")); + } + + /// Withholding is by prefix, not by an enumerated list, so a variable + /// nobody thought of is still withheld. + #[test] + fn an_unanticipated_denied_variable_is_still_withheld() { + let mut h = host(); + h.push(("AWS_SOMETHING_INVENTED_LATER".into(), "x".into())); + h.push(("S3FS_FUTURE_OPTION".into(), "y".into())); + let env = GuestEnvPolicy::default().build_from(&h).unwrap(); + assert!(!names(&env).iter().any(|n| n.contains("INVENTED"))); + assert!(!names(&env).iter().any(|n| n.contains("FUTURE"))); + } + + #[test] + fn an_explicit_value_overrides_inheritance() { + let policy = GuestEnvPolicy { + inherit: true, + explicit: vec!["RUST_LOG=trace".into()], + }; + let env = policy.build_from(&host()).unwrap(); + assert_eq!( + env.iter().find(|(k, _)| k == "RUST_LOG").unwrap().1, + "trace" + ); + } + + /// A deployment that genuinely needs a denied name can still have it, but + /// only by saying so. + #[test] + fn an_explicitly_named_denied_variable_is_allowed_through() { + let policy = GuestEnvPolicy { + inherit: true, + explicit: vec!["AWS_REGION".into()], + }; + let env = policy.build_from(&host()).unwrap(); + assert_eq!( + env.iter().find(|(k, _)| k == "AWS_REGION").unwrap().1, + "eu-west-2" + ); + // ...and nothing else under the prefix came with it. + assert!(!names(&env).contains(&"AWS_SECRET_ACCESS_KEY")); + } + + #[test] + fn explicit_only_inherits_nothing() { + let policy = GuestEnvPolicy::explicit_only(vec!["LANG".into(), "EXTRA=1".into()]); + let env = policy.build_from(&host()).unwrap(); + assert_eq!(names(&env), vec!["EXTRA", "LANG"]); + } + + #[test] + fn explicit_only_with_nothing_named_yields_an_empty_environment() { + assert!(GuestEnvPolicy::explicit_only(vec![]) + .build_from(&host()) + .unwrap() + .is_empty()); + } + + #[test] + fn a_bare_name_unset_on_the_host_is_skipped_not_blanked() { + let policy = GuestEnvPolicy::explicit_only(vec!["NOT_SET_ANYWHERE".into()]); + assert!(policy.build_from(&host()).unwrap().is_empty()); + } + + #[test] + fn an_explicit_empty_value_is_kept() { + let policy = GuestEnvPolicy::explicit_only(vec!["EMPTY=".into()]); + let env = policy.build_from(&host()).unwrap(); + assert_eq!(env, vec![("EMPTY".to_string(), String::new())]); + } + + #[test] + fn a_value_may_contain_equals_signs() { + let policy = GuestEnvPolicy::explicit_only(vec!["OPTS=a=1,b=2".into()]); + let env = policy.build_from(&host()).unwrap(); + assert_eq!(env[0].1, "a=1,b=2"); + } + + #[test] + fn an_empty_variable_name_is_rejected() { + assert!(GuestEnvPolicy::explicit_only(vec!["=value".into()]) + .build_from(&host()) + .is_err()); + assert!(GuestEnvPolicy::explicit_only(vec![String::new()]) + .build_from(&host()) + .is_err()); + } + + /// Two runs with the same inputs must produce the same environment, even + /// though the host's iteration order is not defined. + #[test] + fn output_is_sorted_and_deterministic() { + let mut shuffled = host(); + shuffled.reverse(); + let a = GuestEnvPolicy::default().build_from(&host()).unwrap(); + let b = GuestEnvPolicy::default().build_from(&shuffled).unwrap(); + assert_eq!(a, b); + assert!(a.windows(2).all(|w| w[0].0 < w[1].0)); + } +} diff --git a/runtime/src/flag.rs b/runtime/src/flag.rs new file mode 100644 index 0000000..7d502ec --- /dev/null +++ b/runtime/src/flag.rs @@ -0,0 +1,48 @@ +//! Boolean settings that come from an environment variable. +//! +//! clap's default `bool` handling insists on exactly `true` or `false` when a +//! value arrives through `env`. That is fine on a command line, where the flag +//! is usually written bare, and wrong for a deployment configured by +//! environment: `ENV S3FS_FORCE_PATH_STYLE=1` is the obvious thing to put in a +//! Dockerfile, and it fails at startup with a message about possible values. +//! +//! Inside an enclave that failure is expensive to diagnose — there is no shell +//! to go and check, and the only symptom is an enclave that will not start. + +/// Parse a boolean the way a person writing a Dockerfile would expect. +/// +/// Accepts `1`/`0`, `true`/`false`, `yes`/`no`, `on`/`off`, in any case. An +/// empty value is `false`, matching the shell convention that an unset-looking +/// variable is off. +pub fn parse_bool_flag(s: &str) -> Result { + match s.trim().to_ascii_lowercase().as_str() { + "1" | "true" | "yes" | "on" => Ok(true), + "" | "0" | "false" | "no" | "off" => Ok(false), + other => Err(format!( + "expected a boolean (1/0, true/false, yes/no, on/off), got {other:?}" + )), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn accepts_what_a_dockerfile_would_say() { + for s in ["1", "true", "TRUE", "True", "yes", "on", " 1 "] { + assert_eq!(parse_bool_flag(s), Ok(true), "{s:?}"); + } + for s in ["0", "false", "FALSE", "no", "off", "", " "] { + assert_eq!(parse_bool_flag(s), Ok(false), "{s:?}"); + } + } + + #[test] + fn rejects_anything_ambiguous_with_a_useful_message() { + let err = parse_bool_flag("maybe").unwrap_err(); + assert!(err.contains("maybe")); + assert!(err.contains("1/0")); + assert!(parse_bool_flag("2").is_err()); + } +} diff --git a/runtime/src/guest.rs b/runtime/src/guest.rs new file mode 100644 index 0000000..7467db0 --- /dev/null +++ b/runtime/src/guest.rs @@ -0,0 +1,365 @@ +//! The guest component: fetched from outside the image, measured into PCR16, +//! and locked before anything asks for a key. +//! +//! It used to ship inside the enclave image, which put it under PCR0 and made +//! every guest change an image rebuild. Now the image names *where* the guest +//! is — a key in the roots bucket, covered by PCR0 like the rest of the image's +//! configuration — and the runtime measures *what* it finds there: +//! +//! ```text +//! fetch ──▶ extend PCR16 with sha256(component) ──▶ lock ──▶ boot, and KMS +//! ``` +//! +//! ## Why the object does not have to be trusted +//! +//! Whatever bytes arrive are measured before they can matter. A parent that +//! substitutes the object gets an enclave with a different PCR16: one a key +//! policy pinning the approved guest releases no key to, and one a client +//! pinning that guest refuses. +//! +//! That is also why the order cannot be rearranged. A key released before the +//! lock would go to an enclave whose register could still change, and a +//! register that is never locked appears in no attestation document at all. +//! +//! ## What it does not cover +//! +//! PCR16 means something only beside PCR0. The runtime does the extending, so a +//! runtime someone else wrote can put any value in the register while running +//! anything; PCR0 is what says the runtime that extended it is this one. +//! +//! And measuring a guest is not vetting it. An approved guest can still write +//! whatever it reads to stdout, which the runtime ships to CloudWatch. Approving +//! a guest is trusting it with the data, and nothing here changes that. + +use std::path::PathBuf; +use std::sync::Arc; + +use anyhow::{bail, Context, Result}; +use nitro_nsm::{Nsm, PCR_GUEST, PCR_ZERO}; +use s3fs_core::backend::Backend; +use s3fs_core::FsError; + +/// Larger than any guest this runtime has a use for. +/// +/// The largest example, the SQLite guest, is about 1.3 MiB. The object comes +/// from a store the parent can write to, and memory is the one thing an enclave +/// cannot get more of once it has started. +pub const MAX_GUEST_BYTES: u64 = 64 * 1024 * 1024; + +/// Where the guest component comes from. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum GuestSource { + /// A key in the roots bucket, used verbatim. What an enclave image uses. + Object { key: String }, + /// A local file, for development and tests. Measured exactly as an object + /// is, so a run from a file attests the same PCR16 as one from the store. + Path(PathBuf), +} + +impl std::fmt::Display for GuestSource { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + GuestSource::Object { key } => write!(f, "roots-bucket object {key}"), + GuestSource::Path(path) => write!(f, "file {}", path.display()), + } + } +} + +/// Read the guest component, refusing anything over [`MAX_GUEST_BYTES`]. +pub async fn fetch_guest(source: &GuestSource, roots: &Arc) -> Result> { + fetch_bounded(source, roots, MAX_GUEST_BYTES).await +} + +async fn fetch_bounded( + source: &GuestSource, + roots: &Arc, + limit: u64, +) -> Result> { + // One byte past the limit is read and never more, so an oversized component + // is found out without being held whole. + let bytes = match source { + GuestSource::Object { key } => roots + .get_blob(key, Some(0..limit + 1)) + .await + .map_err(|e| match e { + FsError::NotFound => anyhow::anyhow!( + "no guest component at {key} in the roots bucket; upload the approved \ + guest there before starting the enclave" + ), + other => anyhow::anyhow!("reading the guest component {key}: {other}"), + })? + .body + .to_vec(), + GuestSource::Path(path) => { + use std::io::Read as _; + let file = std::fs::File::open(path) + .with_context(|| format!("opening the guest component {}", path.display()))?; + let mut bytes = Vec::new(); + file.take(limit + 1) + .read_to_end(&mut bytes) + .with_context(|| format!("reading the guest component {}", path.display()))?; + bytes + } + }; + if bytes.len() as u64 > limit { + bail!("the guest component from {source} is over the {limit} byte limit"); + } + if bytes.is_empty() { + bail!("the guest component from {source} is empty"); + } + Ok(bytes) +} + +/// Extend [`PCR_GUEST`] with `sha256(component)` and lock it, returning the +/// value it now holds. +/// +/// Each step is checked against what it should have produced, because each is a +/// place a wrong answer would otherwise pass silently into every attestation +/// this enclave makes: +/// +/// 1. The register must be unlocked and zero. Anything else means something +/// measured into it first, and extending on top would attest a value no +/// client could reproduce from the component. +/// 2. Extending must produce exactly what a client computes from the +/// component, [`nitro_attestation::guest_pcr`]. +/// 3. Lock it. +/// 4. The register must now read locked, with that same value — the device's +/// word for it, not the lock call's. +pub fn measure_guest(nsm: &dyn Nsm, component: &[u8]) -> Result<[u8; 48]> { + let before = nsm + .describe_pcr(PCR_GUEST) + .context("reading PCR16 before measuring the guest")?; + if before.locked { + bail!( + "PCR16 is already locked, so something measured into it before this runtime \ + could. Refusing to start: every attestation would carry a register that names \ + nothing this runtime loaded." + ); + } + if before.value != PCR_ZERO { + bail!( + "PCR16 already holds {}; something extended it before the guest was measured, \ + and extending on top would attest a value no client can reproduce from the \ + component. Refusing to start.", + hex::encode(&before.value) + ); + } + + let expected = nitro_attestation::guest_pcr(component); + let extended = nsm + .extend_pcr(PCR_GUEST, &nitro_attestation::sha256(component)) + .context("extending PCR16 with the guest's hash")?; + if extended != expected { + bail!( + "PCR16 reads {} after extending with the guest's hash, but one extension from \ + zero gives {}. Refusing to start.", + hex::encode(&extended), + hex::encode(expected) + ); + } + + nsm.lock_pcr(PCR_GUEST).context("locking PCR16")?; + + let after = nsm + .describe_pcr(PCR_GUEST) + .context("reading PCR16 after locking it")?; + if !after.locked { + bail!( + "PCR16 still reads unlocked after LockPCR succeeded. An unlocked register appears \ + in no attestation document, so no key policy and no client could see this \ + guest. Refusing to start." + ); + } + if after.value != expected { + bail!( + "PCR16 changed while being locked: it reads {}, expected {}. Refusing to start.", + hex::encode(&after.value), + hex::encode(expected) + ); + } + Ok(expected) +} + +#[cfg(test)] +mod tests { + use super::*; + use nitro_nsm::fake::FakeNsm; + use nitro_nsm::Pcr; + use s3fs_core::backend::memory::MemoryBackend; + use s3fs_core::backend::PutBlobInput; + use std::sync::atomic::{AtomicBool, Ordering}; + + const GUEST: &[u8] = b"\0asm pretend component"; + + #[test] + fn a_guest_is_measured_into_pcr16_and_locked() { + let nsm = FakeNsm::new(); + let measured = measure_guest(&nsm, GUEST).expect("measured"); + assert_eq!( + measured, + nitro_attestation::guest_pcr(GUEST), + "the value a client computes from the component" + ); + let pcr = nsm.describe_pcr(PCR_GUEST).unwrap(); + assert!(pcr.locked); + assert_eq!(pcr.value, measured.to_vec()); + } + + /// The runtime and every client compute this register with two copies of + /// the arithmetic, on either side of a crate boundary that exists on + /// purpose. This is what keeps them the same arithmetic. + #[test] + fn the_device_side_and_the_verifier_side_agree() { + assert_eq!(PCR_GUEST as u32, nitro_attestation::PCR_GUEST); + let digest = nitro_attestation::sha256(GUEST); + let cases: [&[u8]; 4] = [b"", b"abc", &[0xff; 32], &digest]; + for data in cases { + assert_eq!( + nitro_nsm::pcr_extend(&PCR_ZERO, data), + nitro_attestation::pcr_after_one_extend(data).to_vec() + ); + } + } + + /// Something extended the register first. + #[test] + fn a_register_already_extended_is_refused() { + let nsm = FakeNsm::new(); + nsm.extend_pcr(PCR_GUEST, b"someone else").unwrap(); + let err = measure_guest(&nsm, GUEST).unwrap_err(); + assert!(format!("{err:#}").contains("already holds"), "{err:#}"); + } + + #[test] + fn a_register_already_locked_is_refused() { + let nsm = FakeNsm::new(); + nsm.lock_pcr(PCR_GUEST).unwrap(); + let err = measure_guest(&nsm, GUEST).unwrap_err(); + assert!(format!("{err:#}").contains("already locked"), "{err:#}"); + } + + /// A device that misreports, one lie at a time. + #[derive(Debug, Default)] + struct Misreporting { + inner: FakeNsm, + wrong_extend: AtomicBool, + lock_ignored: AtomicBool, + } + + impl Nsm for Misreporting { + fn get_random(&self, buf: &mut [u8]) -> Result<()> { + self.inner.get_random(buf) + } + fn attest(&self, request: &nitro_nsm::AttestationRequest) -> Result> { + self.inner.attest(request) + } + fn describe_pcr(&self, index: u16) -> Result { + self.inner.describe_pcr(index) + } + fn extend_pcr(&self, index: u16, data: &[u8]) -> Result> { + let value = self.inner.extend_pcr(index, data)?; + if self.wrong_extend.load(Ordering::SeqCst) { + return Ok(vec![0xee; 48]); + } + Ok(value) + } + fn lock_pcr(&self, index: u16) -> Result<()> { + if self.lock_ignored.load(Ordering::SeqCst) { + return Ok(()); + } + self.inner.lock_pcr(index) + } + fn describe(&self) -> String { + "misreporting test NSM".into() + } + } + + #[test] + fn an_extension_that_did_not_produce_the_measurement_is_refused() { + let nsm = Misreporting::default(); + nsm.wrong_extend.store(true, Ordering::SeqCst); + let err = measure_guest(&nsm, GUEST).unwrap_err(); + assert!(format!("{err:#}").contains("after extending"), "{err:#}"); + assert!( + !nsm.describe_pcr(PCR_GUEST).unwrap().locked, + "a bad measurement was locked in" + ); + } + + /// The lock call returning is not the device saying the register is + /// locked. An unlocked register is in no document, so this is the + /// difference between a guest the key policy can see and one it cannot. + #[test] + fn a_lock_that_did_not_take_is_refused() { + let nsm = Misreporting::default(); + nsm.lock_ignored.store(true, Ordering::SeqCst); + let err = measure_guest(&nsm, GUEST).unwrap_err(); + assert!(format!("{err:#}").contains("unlocked"), "{err:#}"); + } + + async fn roots_with(key: &str, body: &[u8]) -> Arc { + let roots: Arc = Arc::new(MemoryBackend::new()); + roots + .put_blob(PutBlobInput::new(key, body.to_vec().into())) + .await + .expect("write"); + roots + } + + #[tokio::test] + async fn a_guest_object_is_read_from_the_roots_bucket() { + let roots = roots_with("guest/guest.wasm", GUEST).await; + let source = GuestSource::Object { + key: "guest/guest.wasm".into(), + }; + assert_eq!(fetch_guest(&source, &roots).await.unwrap(), GUEST); + } + + #[tokio::test] + async fn a_missing_guest_object_says_what_to_do() { + let roots: Arc = Arc::new(MemoryBackend::new()); + let source = GuestSource::Object { + key: "guest/guest.wasm".into(), + }; + let err = fetch_guest(&source, &roots).await.unwrap_err(); + assert!(format!("{err:#}").contains("upload"), "{err:#}"); + } + + #[tokio::test] + async fn an_oversized_guest_object_is_refused() { + let roots = roots_with("big", &[7u8; 17]).await; + let source = GuestSource::Object { key: "big".into() }; + let err = fetch_bounded(&source, &roots, 16).await.unwrap_err(); + assert!(format!("{err:#}").contains("limit"), "{err:#}"); + assert_eq!( + fetch_bounded(&source, &roots, 17).await.unwrap(), + vec![7u8; 17], + "exactly at the limit is allowed" + ); + } + + #[tokio::test] + async fn a_guest_file_is_bounded_the_same_way() { + let path = std::env::temp_dir().join(format!("guest-bound-{}.wasm", std::process::id())); + std::fs::write(&path, [7u8; 17]).unwrap(); + let roots: Arc = Arc::new(MemoryBackend::new()); + let source = GuestSource::Path(path.clone()); + + let refused = fetch_bounded(&source, &roots, 16).await; + let accepted = fetch_bounded(&source, &roots, 17).await; + let _ = std::fs::remove_file(&path); + + assert!(refused.is_err()); + assert_eq!(accepted.unwrap(), vec![7u8; 17]); + } + + #[tokio::test] + async fn an_empty_guest_is_refused() { + let roots = roots_with("empty", b"").await; + let source = GuestSource::Object { + key: "empty".into(), + }; + let err = fetch_guest(&source, &roots).await.unwrap_err(); + assert!(format!("{err:#}").contains("empty"), "{err:#}"); + } +} diff --git a/runtime/src/guest_io.rs b/runtime/src/guest_io.rs new file mode 100644 index 0000000..8e91d47 --- /dev/null +++ b/runtime/src/guest_io.rs @@ -0,0 +1,1273 @@ +//! Guest stdout and stderr, framed into bounded records and handed to the +//! runtime's own logging pipeline. +//! +//! ```text +//! guest write ──▶ GuestLogOutput ──▶ bounded mpsc ──▶ GuestLogCollector +//! (line framing, (drop when (owns the sink) +//! truncation) full) │ +//! ▼ +//! GuestLogSink +//! └── TracingLogSink +//! ``` +//! +//! # Guest output is attacker-controlled data +//! +//! Everything that arrives here was chosen by the guest. It is **not** a +//! runtime audit event and must never be read as one. A guest can emit +//! anything it likes, including text shaped exactly like this runtime's own +//! log lines, so the two are kept apart by the tracing *target* +//! ([`GUEST_LOG_TARGET`]) rather than by anything in the message. A filter on +//! that target separates trusted runtime events from untrusted guest output; +//! nothing in the message body is load-bearing. +//! +//! Guest text is never interpolated into an event name or a target, only into +//! a field value, so a guest cannot forge an event that looks like a different +//! kind of event. +//! +//! # Best-effort, and lossy under pressure +//! +//! A guest write must never wait on a log destination — not on this process's +//! collector, not on the parent instance, and not on CloudWatch. So the queue +//! is bounded and **records are dropped when it is full**. The guest sees a +//! successful write either way: logging congestion is a host concern and must +//! not become a guest-visible failure, let alone a stall. +//! +//! What that buys, and what it costs: +//! +//! - The guest cannot make the enclave allocate without bound by writing. Live +//! memory is capped by the queue ([`LOG_QUEUE_CAPACITY`] records) plus one +//! partial line per open stream ([`MAX_LOG_RECORD_BYTES`] each). +//! - A stalled or slow sink cannot stall a guest. +//! - There is **no delivery guarantee**. Output is lost when the queue fills, +//! and output still queued is lost if the enclave stops abruptly. Drops are +//! counted and reported in aggregate, never one warning per dropped write — +//! which would hand a guest an amplification primitive against the very log +//! it is congesting. +//! +//! Delivery beyond the console is a separate concern behind [`GuestLogSink`], +//! and nothing here should be treated as an authenticated record of what a +//! guest did: the content is guest-chosen whatever carries it. + +pub mod cloudwatch; + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; +use std::time::Duration; + +use async_trait::async_trait; +use bytes::Bytes; +use tokio::sync::{mpsc, Notify}; +use wasmtime_wasi::cli::{IsTerminal, StdoutStream}; +use wasmtime_wasi::p2::{OutputStream, Pollable, StreamResult}; + +/// The tracing target every guest-produced event carries. +/// +/// The one thing separating untrusted guest output from the runtime's own +/// events. A subscriber can route or drop this target wholesale; a guest +/// cannot change it, because it is never derived from anything the guest +/// wrote. +pub const GUEST_LOG_TARGET: &str = "guest"; + +/// Records held between the guest and the collector before output is dropped. +/// +/// Bounds live memory together with [`MAX_LOG_RECORD_BYTES`]: at most this +/// many records, each at most that large. +pub const LOG_QUEUE_CAPACITY: usize = 1024; + +/// Ceiling on one emitted record. A longer line is cut here and marked +/// truncated rather than retained. +pub const MAX_LOG_RECORD_BYTES: usize = 16 * 1024; + +/// The permit `check_write` advertises. A guest that respects it writes at +/// most this much per call; one that ignores it is still safe, because what +/// this module *retains* is bounded by [`MAX_LOG_RECORD_BYTES`] regardless of +/// how much arrives at once. +pub const MAX_WRITE_CHUNK_BYTES: usize = 16 * 1024; + +/// How long a graceful shutdown will wait for queued records to drain. +/// +/// Fixed and short: losing the tail of a log is a nuisance, and an enclave +/// that will not stop is an outage. +const DRAIN_DEADLINE: Duration = Duration::from_secs(2); + +/// How often the collector reports dropped output, if any was dropped. +const DROP_REPORT_INTERVAL: Duration = Duration::from_secs(30); + +/// Which of the guest's two output streams a record came from. +/// +/// Kept distinct all the way to the sink. A guest that writes a diagnostic to +/// stderr and data to stdout is making a distinction, and merging them would +/// destroy it — but note that this says only *which file descriptor* was +/// written to. It is not a severity, and none is inferred from the text. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum GuestStream { + Stdout, + Stderr, +} + +impl GuestStream { + /// The value that appears in the `guest_stream` field. + pub fn as_str(self) -> &'static str { + match self { + GuestStream::Stdout => "stdout", + GuestStream::Stderr => "stderr", + } + } +} + +/// One framed line of guest output. +/// +/// Deliberately small. Metadata that would need request- or tenant-scope — +/// which tenant wrote this, which request it belonged to — is *not* here: it +/// is not available where the WASI context is built without widening that +/// seam, and inventing a path for it was out of scope for this change. When it +/// arrives it belongs on this struct, filled in by the collector. +/// +/// Never carries credentials, tokens, challenges, request bodies or raw +/// configuration, because nothing on this path has access to any of them: the +/// only input is the bytes the guest itself wrote. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GuestLogRecord { + pub stream: GuestStream, + /// The line, minus its terminator. Invalid UTF-8 has been replaced + /// lossily; empty is a legitimate value, because the guest wrote a blank + /// line and that is worth keeping. + pub message: String, + /// The line was longer than [`MAX_LOG_RECORD_BYTES`] and what is here is a + /// prefix. Explicit, so nothing downstream mistakes a cut line for what + /// the guest actually wrote. + pub truncated: bool, +} + +/// Where framed guest output finally goes. +/// +/// Async, and owned by the collector task — **never** called from a WASI +/// write. That is the whole point of the split: an implementation may take as +/// long as it likes without a guest ever waiting on it. +#[async_trait] +pub trait GuestLogSink: Send + Sync { + async fn emit(&self, record: GuestLogRecord); +} + +/// Emits guest output as structured tracing events under [`GUEST_LOG_TARGET`]. +/// +/// stderr is emitted at `warn` and stdout at `info`. That is a statement about +/// *which stream was written to*, not about what the text means: severity is +/// never inferred from guest-controlled content, because a guest could then +/// choose its own log level. +#[derive(Debug, Default, Clone, Copy)] +pub struct TracingLogSink; + +#[async_trait] +impl GuestLogSink for TracingLogSink { + async fn emit(&self, record: GuestLogRecord) { + // `guest_message`, not `message`, and recorded as a `&str` rather than + // through `%`. Both matter, and the enclave console is what showed it: + // + // `message` is the field name `tracing` gives an event's own text, so + // using it put guest bytes *bare and unquoted* at the end of the line, + // in the structured part, where a guest writing `truncated=true` would + // render as a field nobody set. A `&str` field is printed quoted and + // escaped, so guest text can only ever be one field's value — + // `%` would have formatted it raw again and reopened the same hole. + // + // It is still only ever a field *value*: never the event name, never + // the target. + match record.stream { + GuestStream::Stdout => tracing::info!( + target: GUEST_LOG_TARGET, + guest_stream = GuestStream::Stdout.as_str(), + truncated = record.truncated, + guest_message = record.message.as_str(), + "guest output" + ), + GuestStream::Stderr => tracing::warn!( + target: GUEST_LOG_TARGET, + guest_stream = GuestStream::Stderr.as_str(), + truncated = record.truncated, + guest_message = record.message.as_str(), + "guest output" + ), + } + } +} + +/// Sends every record to several sinks. +/// +/// The console keeps working when a relay is configured. Guest output on the +/// enclave console is often the only thing available while a deployment is +/// being brought up — and the relay is precisely the part that may not work +/// yet — so forwarding is additive rather than a replacement. +/// +/// Sinks run in order and a slow one delays the ones after it. That is the +/// collector's own budget to spend: it is already off the guest's path, and +/// none of it reaches a request. +pub struct FanOutSink { + sinks: Vec>, +} + +impl FanOutSink { + pub fn new(sinks: Vec>) -> Self { + FanOutSink { sinks } + } +} + +impl std::fmt::Debug for FanOutSink { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("FanOutSink") + .field("sinks", &self.sinks.len()) + .finish_non_exhaustive() + } +} + +#[async_trait] +impl GuestLogSink for FanOutSink { + async fn emit(&self, record: GuestLogRecord) { + // Cloned per sink because each takes ownership; a record is one line, + // capped at `MAX_LOG_RECORD_BYTES`, and there are two sinks. + for sink in &self.sinks { + sink.emit(record.clone()).await; + } + } +} + +/// Output that was lost, counted rather than logged one line at a time. +/// +/// Two ways to lose output, and both belong here: a whole record dropped +/// because the queue was full, and the bytes past [`MAX_LOG_RECORD_BYTES`] cut +/// from a line that was too long. `records` counts only the first — a +/// truncated line still arrives, marked — while `bytes` counts what was +/// actually lost either way. +#[derive(Debug, Default)] +struct DropCounters { + records: AtomicU64, + bytes: AtomicU64, +} + +impl DropCounters { + fn record(&self, bytes: usize) { + self.records.fetch_add(1, Ordering::Relaxed); + self.bytes.fetch_add(bytes as u64, Ordering::Relaxed); + } + + fn snapshot(&self) -> (u64, u64) { + ( + self.records.load(Ordering::Relaxed), + self.bytes.load(Ordering::Relaxed), + ) + } +} + +/// The handle a [`crate::run::GuestEnvironment`] holds, and the factory every +/// guest instance takes its streams from. +/// +/// Cheap to clone. Every stream it hands out feeds the same bounded queue and +/// the same collector, however many instances exist. +#[derive(Clone)] +pub struct GuestLogs { + tx: mpsc::Sender, + drops: Arc, +} + +impl std::fmt::Debug for GuestLogs { + /// Never prints queued records: they are guest-controlled text, and a + /// `Debug` that included them would put it in any log line that formatted + /// a struct holding one. + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + let (records, bytes) = self.drops.snapshot(); + f.debug_struct("GuestLogs") + .field("queue_capacity", &LOG_QUEUE_CAPACITY) + .field("dropped_records", &records) + .field("dropped_bytes", &bytes) + .finish_non_exhaustive() + } +} + +impl GuestLogs { + /// The guest's stdout, as something [`wasmtime_wasi::WasiCtxBuilder`] + /// accepts. + pub fn stdout(&self) -> GuestLogStdio { + GuestLogStdio { + stream: GuestStream::Stdout, + tx: self.tx.clone(), + drops: self.drops.clone(), + } + } + + /// The guest's stderr. `WasiCtxBuilder::stderr` also takes a + /// [`StdoutStream`]; the name is wasmtime's, and the tagging here is what + /// keeps the two apart. + pub fn stderr(&self) -> GuestLogStdio { + GuestLogStdio { + stream: GuestStream::Stderr, + tx: self.tx.clone(), + drops: self.drops.clone(), + } + } + + /// Whole records dropped because the queue was full, since start. + /// + /// Does not count truncated lines: those still arrive, carrying + /// [`GuestLogRecord::truncated`]. + pub fn dropped_records(&self) -> u64 { + self.drops.snapshot().0 + } + + /// Bytes of guest output lost since start, whether to a full queue or to + /// the per-record cap. + pub fn dropped_bytes(&self) -> u64 { + self.drops.snapshot().1 + } +} + +/// Owns the collector task, so its lifetime is explicit rather than resting on +/// a sender that happens to still be alive somewhere. +/// +/// Dropping this handle **detaches** the task rather than stopping it: the +/// collector keeps running for as long as any [`GuestLogs`] can still send, so +/// a caller that forgets to shut down loses no output. Only +/// [`GuestLogCollector::shutdown`] drains and stops. +pub struct GuestLogCollector { + task: tokio::task::JoinHandle<()>, + stop: Arc, +} + +impl std::fmt::Debug for GuestLogCollector { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("GuestLogCollector") + .field("finished", &self.task.is_finished()) + .finish_non_exhaustive() + } +} + +impl GuestLogCollector { + /// Drain what is queued and stop, within [`DRAIN_DEADLINE`]. + /// + /// Bounded on purpose. Shutdown must not become dependent on a sink that + /// has stopped making progress, so a collector that overruns is abandoned + /// and the fact is logged. + pub async fn shutdown(mut self) { + // `notify_one`, not `notify_waiters`: it stores a permit when the task + // has not yet registered, so a shutdown that races the collector's + // first poll still stops it instead of waiting out the deadline. + self.stop.notify_one(); + match tokio::time::timeout(DRAIN_DEADLINE, &mut self.task).await { + Ok(Ok(())) => {} + Ok(Err(e)) => tracing::warn!(error = %e, "the guest log collector ended badly"), + Err(_) => { + tracing::warn!( + deadline = ?DRAIN_DEADLINE, + "the guest log collector did not drain in time; dropping queued guest output" + ); + self.task.abort(); + } + } + } +} + +/// Start the collector and return the handle guests write through. +/// +/// Call this **before** any guest can run: the returned [`GuestLogs`] is what +/// builds a guest's streams, so no output can exist before the task that +/// consumes it. +/// +/// The task ends when [`GuestLogCollector::shutdown`] is called, or when every +/// [`GuestLogs`] has been dropped — but do not rely on the latter for +/// shutdown, which is exactly why the handle is returned rather than detached. +pub fn start(sink: Arc) -> (GuestLogs, GuestLogCollector) { + let (tx, mut rx) = mpsc::channel::(LOG_QUEUE_CAPACITY); + let stop = Arc::new(Notify::new()); + let drops = Arc::new(DropCounters::default()); + + let task = tokio::spawn({ + let drops = drops.clone(); + let stop = stop.clone(); + async move { + let stopping = stop.notified(); + tokio::pin!(stopping); + let mut report = tokio::time::interval(DROP_REPORT_INTERVAL); + report.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + // The first tick completes immediately; nothing has been dropped + // yet, so let it pass rather than reporting a zero. + report.tick().await; + let mut reported = (0u64, 0u64); + + loop { + // Deliberately *not* `biased`. Polling records first starves the + // other two branches whenever the queue is never empty — + // which is exactly the sustained overload the drop report + // exists to describe, so the report would go silent precisely + // when it matters. Nothing is lost by letting shutdown win a + // race: its branch drains the queue explicitly. + tokio::select! { + received = rx.recv() => match received { + Some(record) => sink.emit(record).await, + // Every sender is gone, so nothing more can arrive. + None => break, + }, + _ = report.tick() => { + reported = report_drops(&drops, reported); + } + _ = &mut stopping => { + // Whatever is already queued, then stop. Anything a + // still-running guest writes after this is dropped, + // which is the documented behaviour of a shutdown. + while let Ok(record) = rx.try_recv() { + sink.emit(record).await; + } + break; + } + } + } + report_drops(&drops, reported); + } + }); + + (GuestLogs { tx, drops }, GuestLogCollector { task, stop }) +} + +/// One aggregate warning per interval, never one per dropped write. +/// +/// Returns the new high-water mark so the next report covers only what has +/// been dropped since. +fn report_drops(drops: &DropCounters, since: (u64, u64)) -> (u64, u64) { + let (records, bytes) = drops.snapshot(); + // Either kind of loss is worth reporting: a guest can lose everything it + // writes to truncation alone without a single record being dropped. + if records > since.0 || bytes > since.1 { + tracing::warn!( + dropped_records = records - since.0, + lost_bytes = bytes - since.1, + total_dropped_records = records, + "guest output was lost: the log queue was full, or lines exceeded the record cap" + ); + } + (records, bytes) +} + +/// A guest stdio stream, as wasmtime's CLI layer wants it. +/// +/// This is the *factory*. `p2_stream` and `async_stream` are called to make a +/// fresh object each time the guest asks for its stdout or stderr, and each of +/// those objects carries its own partial-line buffer — see [`LineFramer`]. +#[derive(Clone)] +pub struct GuestLogStdio { + stream: GuestStream, + tx: mpsc::Sender, + drops: Arc, +} + +impl std::fmt::Debug for GuestLogStdio { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("GuestLogStdio") + .field("stream", &self.stream) + .finish_non_exhaustive() + } +} + +impl GuestLogStdio { + fn framer(&self) -> LineFramer { + LineFramer::new(self.stream, self.tx.clone(), self.drops.clone()) + } +} + +impl IsTerminal for GuestLogStdio { + /// Never a terminal. There is no console in an enclave, and a guest told + /// otherwise may enable colour escapes or line-editing behaviour that + /// makes its output worse to read and no easier to parse. + fn is_terminal(&self) -> bool { + false + } +} + +impl StdoutStream for GuestLogStdio { + /// Implemented directly rather than through the default + /// `AsyncWriteStream` adapter: writes here are pure local framing plus a + /// non-blocking enqueue, so there is no readiness to model and no reason + /// to pay for an intermediate buffer. + fn p2_stream(&self) -> Box { + Box::new(GuestLogOutput { + framer: self.framer(), + }) + } + + /// Present because the trait requires it. It frames identically, so a host + /// that reaches for the `AsyncWrite` shape gets the same records. + fn async_stream(&self) -> Box { + Box::new(GuestLogWriter { + framer: self.framer(), + }) + } +} + +/// The `wasi:io/streams` output stream a guest actually writes into. +struct GuestLogOutput { + framer: LineFramer, +} + +#[async_trait] +impl Pollable for GuestLogOutput { + /// Always ready. There is no destination to wait for — a write frames + /// locally and enqueues or drops — so a guest never blocks on logging. + async fn ready(&mut self) {} +} + +impl OutputStream for GuestLogOutput { + /// Constant, and never zero. Returning zero would make a guest poll for + /// capacity that a full queue will never grant, which is the stall this + /// design exists to avoid. + fn check_write(&mut self) -> StreamResult { + Ok(MAX_WRITE_CHUNK_BYTES) + } + + /// Frames and enqueues, and always succeeds. + /// + /// A write that is dropped for congestion still reports success: the guest + /// asked to write to its own stdout, and whether the *host* keeps that is + /// not something the guest did wrong. Errors are reserved for the stream + /// being unusable, which this one never is. + fn write(&mut self, bytes: Bytes) -> StreamResult<()> { + self.framer.push(&bytes); + Ok(()) + } + + /// Local framing only. + /// + /// Deliberately **not** a flush of the partial line. A guest that writes + /// `"foo"`, flushes, then writes `"bar\n"` wrote one line, and emitting + /// `foo` here would split it. There is nothing else to flush: a completed + /// line is enqueued the moment its terminator arrives, and the tail is + /// emitted when the stream is dropped. + fn flush(&mut self) -> StreamResult<()> { + Ok(()) + } +} + +/// The `AsyncWrite` shape of the same thing. +struct GuestLogWriter { + framer: LineFramer, +} + +impl tokio::io::AsyncWrite for GuestLogWriter { + fn poll_write( + mut self: std::pin::Pin<&mut Self>, + _cx: &mut std::task::Context<'_>, + buf: &[u8], + ) -> std::task::Poll> { + self.framer.push(buf); + std::task::Poll::Ready(Ok(buf.len())) + } + + fn poll_flush( + self: std::pin::Pin<&mut Self>, + _cx: &mut std::task::Context<'_>, + ) -> std::task::Poll> { + std::task::Poll::Ready(Ok(())) + } + + fn poll_shutdown( + mut self: std::pin::Pin<&mut Self>, + _cx: &mut std::task::Context<'_>, + ) -> std::task::Poll> { + self.framer.finish(); + std::task::Poll::Ready(Ok(())) + } +} + +/// Turns a byte stream into whole lines, one stream object at a time. +/// +/// **Per stream object, deliberately.** `StdoutStream::p2_stream` hands out a +/// fresh object every time the guest asks for its stdout, and a guest may hold +/// several at once. Partial-line state kept in the shared factory would splice +/// two of those together into a line neither of them wrote; keeping it here +/// means an unfinished line belongs to exactly the stream that wrote it, and +/// goes away with it. +/// +/// It also keeps the memory bound simple. The only thing retained between +/// writes is one partial line, capped at [`MAX_LOG_RECORD_BYTES`], so a guest +/// writing a gigabyte without a newline costs sixteen kilobytes and a +/// truncation flag. +struct LineFramer { + stream: GuestStream, + tx: mpsc::Sender, + drops: Arc, + /// The line so far, never longer than [`MAX_LOG_RECORD_BYTES`]. + partial: Vec, + /// The current line already exceeded the cap: keep the prefix, discard the + /// rest until a terminator, and mark the record. + overflowed: bool, +} + +impl LineFramer { + fn new( + stream: GuestStream, + tx: mpsc::Sender, + drops: Arc, + ) -> Self { + LineFramer { + stream, + tx, + drops, + partial: Vec::new(), + overflowed: false, + } + } + + /// Absorb a write, emitting a record for every completed line in it. + fn push(&mut self, mut bytes: &[u8]) { + while let Some(at) = bytes.iter().position(|&b| b == b'\n') { + self.extend(&bytes[..at]); + self.emit(); + bytes = &bytes[at + 1..]; + } + self.extend(bytes); + } + + /// Append what still fits, and account for what does not. + /// + /// Bytes past the cap are counted as dropped and discarded here rather + /// than buffered, which is what makes a write of any size cost a bounded + /// amount of memory. + fn extend(&mut self, bytes: &[u8]) { + if bytes.is_empty() { + return; + } + let room = MAX_LOG_RECORD_BYTES.saturating_sub(self.partial.len()); + if room == 0 { + self.overflowed = true; + self.drops + .bytes + .fetch_add(bytes.len() as u64, Ordering::Relaxed); + return; + } + if bytes.len() > room { + self.partial.extend_from_slice(&bytes[..room]); + self.overflowed = true; + self.drops + .bytes + .fetch_add((bytes.len() - room) as u64, Ordering::Relaxed); + } else { + self.partial.extend_from_slice(bytes); + } + } + + /// Emit the buffered line and start a new one. + fn emit(&mut self) { + let mut line = std::mem::take(&mut self.partial); + let truncated = std::mem::replace(&mut self.overflowed, false); + // `\r\n` is normalised to `\n`, so a guest built against Windows + // conventions does not leave a stray carriage return in every record. + // Only the terminator's own `\r` is removed; one in the middle of a + // line is the guest's business. + if line.last() == Some(&b'\r') { + line.pop(); + } + // Lossy rather than rejected: invalid UTF-8 is a thing a guest can + // simply write, and losing the whole line — or worse, panicking — over + // a stray byte would be a guest-triggered failure. + let message = String::from_utf8_lossy(&line).into_owned(); + let record = GuestLogRecord { + stream: self.stream, + message, + truncated, + }; + // `try_send`, never `send`: this is called from a guest write, and + // awaiting here is exactly the stall this module exists to prevent. + if let Err(e) = self.tx.try_send(record) { + let dropped = match e { + mpsc::error::TrySendError::Full(r) => r, + mpsc::error::TrySendError::Closed(r) => r, + }; + self.drops.record(dropped.message.len()); + } + } + + /// Emit a final unterminated line, if there is one. + /// + /// A guest that writes `"done"` and exits without a newline still said + /// something, and dropping it would make the last thing a failing guest + /// reported the most likely thing to be lost. + /// + /// Emitted when the stream object is dropped, which is when the guest's + /// stdout resource goes — shortly after the request that wrote it, not at + /// some later flush. Bounded regardless: one partial line per stream, + /// capped at [`MAX_LOG_RECORD_BYTES`]. + fn finish(&mut self) { + if !self.partial.is_empty() || self.overflowed { + self.emit(); + } + } +} + +impl Drop for LineFramer { + /// Closing the stream is what ends the last line. There is no explicit + /// close in `wasi:io/streams`; the resource being dropped is the signal. + fn drop(&mut self) { + self.finish(); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// A `GuestLogs` whose queue this test owns, so records can be read back + /// without a collector in the way. + fn harness(capacity: usize) -> (GuestLogs, mpsc::Receiver) { + let (tx, rx) = mpsc::channel(capacity); + ( + GuestLogs { + tx, + drops: Arc::new(DropCounters::default()), + }, + rx, + ) + } + + /// Everything queued, in order. Takes the stream by value so its `Drop` + /// runs first — that is what ends an unterminated final line. + fn drain( + stream: Box, + rx: &mut mpsc::Receiver, + ) -> Vec { + drop(stream); + let mut out = Vec::new(); + while let Ok(record) = rx.try_recv() { + out.push(record); + } + out + } + + fn write(stream: &mut Box, bytes: &'static [u8]) { + stream.write(Bytes::from_static(bytes)).expect("write"); + } + + fn messages(records: &[GuestLogRecord]) -> Vec<&str> { + records.iter().map(|r| r.message.as_str()).collect() + } + + /// The distinction the whole module exists to preserve. A guest writing a + /// diagnostic to stderr and data to stdout means two different things, and + /// a pipeline that merged them would destroy that before anything could + /// act on it. + #[test] + fn stdout_and_stderr_are_tagged_apart() { + let (logs, mut rx) = harness(8); + let mut out = logs.stdout().p2_stream(); + let mut err = logs.stderr().p2_stream(); + write(&mut out, b"to stdout\n"); + write(&mut err, b"to stderr\n"); + drop(out); + drop(err); + + let mut seen: Vec<(GuestStream, String)> = Vec::new(); + while let Ok(r) = rx.try_recv() { + seen.push((r.stream, r.message)); + } + assert!( + seen.contains(&(GuestStream::Stdout, "to stdout".into())), + "{seen:?}" + ); + assert!( + seen.contains(&(GuestStream::Stderr, "to stderr".into())), + "{seen:?}" + ); + } + + #[test] + fn one_write_one_line() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + write(&mut s, b"hello\n"); + let records = drain(s, &mut rx); + assert_eq!(messages(&records), vec!["hello"]); + assert!(!records[0].truncated); + } + + /// A guest is under no obligation to write whole lines. `print!` followed + /// by `println!` is two writes and one line, and reporting it as two would + /// misrepresent what the guest said. + #[test] + fn a_line_split_across_writes_is_joined() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + write(&mut s, b"one "); + write(&mut s, b"two "); + write(&mut s, b"three\n"); + assert_eq!(messages(&drain(s, &mut rx)), vec!["one two three"]); + } + + #[test] + fn many_lines_in_one_write() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + write(&mut s, b"a\nb\nc\n"); + assert_eq!(messages(&drain(s, &mut rx)), vec!["a", "b", "c"]); + } + + /// A blank line is something the guest wrote, and spacing can be the whole + /// meaning of it. Dropping empties would also silently renumber anything + /// counting lines. + #[test] + fn an_empty_line_is_still_a_record() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + write(&mut s, b"a\n\nb\n"); + assert_eq!(messages(&drain(s, &mut rx)), vec!["a", "", "b"]); + } + + #[test] + fn crlf_is_normalised() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + write(&mut s, b"windows\r\nunix\n"); + assert_eq!(messages(&drain(s, &mut rx)), vec!["windows", "unix"]); + } + + /// Only the terminator's own carriage return is removed. One in the middle + /// of a line is content, and rewriting content would be lying about what + /// the guest wrote. + #[test] + fn an_interior_carriage_return_is_left_alone() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + write(&mut s, b"a\rb\n"); + assert_eq!(messages(&drain(s, &mut rx)), vec!["a\rb"]); + } + + /// A guest can write any bytes it likes. Losing the line — or panicking + /// over it — would hand the guest a way to suppress logging, or worse. + #[test] + fn invalid_utf8_is_replaced_not_dropped() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + write(&mut s, b"bad \xff\xfe end\n"); + let records = drain(s, &mut rx); + assert_eq!(records.len(), 1); + assert!(records[0].message.starts_with("bad "), "{:?}", records[0]); + assert!(records[0].message.ends_with(" end"), "{:?}", records[0]); + assert!(records[0].message.contains('\u{fffd}'), "{:?}", records[0]); + } + + /// The last thing a failing guest reports is the thing most worth keeping, + /// and it is exactly the thing with no newline after it. + #[test] + fn an_unterminated_line_is_emitted_when_the_stream_closes() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + write(&mut s, b"no newline here"); + assert!( + rx.try_recv().is_err(), + "nothing should be emitted before close" + ); + assert_eq!(messages(&drain(s, &mut rx)), vec!["no newline here"]); + } + + /// Closing after a clean line must not invent an empty record. + #[test] + fn closing_after_a_complete_line_emits_nothing_extra() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + write(&mut s, b"done\n"); + assert_eq!(messages(&drain(s, &mut rx)), vec!["done"]); + } + + /// Flushing mid-line must not split it. A guest that writes `"foo"`, + /// flushes, then writes `"bar\n"` wrote one line, and `flush` is not a + /// statement that the line is over. + #[test] + fn flush_does_not_split_a_line() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + write(&mut s, b"foo"); + s.flush().expect("flush"); + assert!(rx.try_recv().is_err(), "flush must not emit a partial line"); + write(&mut s, b"bar\n"); + assert_eq!(messages(&drain(s, &mut rx)), vec!["foobar"]); + } + + #[test] + fn an_oversized_line_is_cut_and_marked() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + let huge = vec![b'x'; MAX_LOG_RECORD_BYTES * 3]; + s.write(Bytes::from(huge)).expect("write"); + write(&mut s, b"\ntail\n"); + let records = drain(s, &mut rx); + assert_eq!(records.len(), 2); + assert_eq!(records[0].message.len(), MAX_LOG_RECORD_BYTES); + assert!(records[0].truncated, "an cut line must say so"); + // The cut ends with the line, so the next one starts clean rather than + // inheriting the overflow flag. + assert_eq!(records[1].message, "tail"); + assert!(!records[1].truncated); + } + + /// The bound that stops a guest turning `println!` into an allocator. A + /// single enormous write must cost a fixed amount of retained memory, not + /// a proportional one. + #[test] + fn an_oversized_write_does_not_allocate_without_bound() { + let (tx, _rx) = mpsc::channel(8); + let mut framer = + LineFramer::new(GuestStream::Stdout, tx, Arc::new(DropCounters::default())); + framer.push(&vec![b'x'; 4 * 1024 * 1024]); + assert!( + framer.partial.len() <= MAX_LOG_RECORD_BYTES, + "retained {} bytes for a 4 MiB write", + framer.partial.len() + ); + assert!(framer.overflowed); + } + + /// The requirement that outranks delivery: a guest write returns, whatever + /// the state of the log queue. A stalled sink is a host problem and must + /// never become a guest stall or a guest-visible error. + #[test] + fn a_full_queue_never_blocks_or_fails_a_guest_write() { + let (logs, _rx) = harness(1); + let mut s = logs.stdout().p2_stream(); + for _ in 0..10_000 { + // Never `Err`, and never awaits: if this test hangs or fails, the + // design has been broken. + s.write(Bytes::from_static(b"line\n")) + .expect("guest write must succeed"); + assert_eq!(s.check_write().expect("permit"), MAX_WRITE_CHUNK_BYTES); + } + } + + #[test] + fn dropped_output_is_counted() { + let (logs, _rx) = harness(1); + let mut s = logs.stdout().p2_stream(); + for _ in 0..100 { + write(&mut s, b"line\n"); + } + drop(s); + assert!( + logs.dropped_records() > 0, + "a queue of one and a hundred lines must drop something" + ); + assert!(logs.dropped_bytes() > 0); + } + + /// Truncated bytes are accounted too, so the drop report reflects how much + /// output was actually lost rather than only how many records. + #[test] + fn truncated_bytes_are_accounted() { + let (logs, mut rx) = harness(8); + let mut s = logs.stdout().p2_stream(); + let overshoot = MAX_LOG_RECORD_BYTES * 2; + s.write(Bytes::from(vec![b'x'; overshoot])).expect("write"); + let _ = drain(s, &mut rx); + assert_eq!( + logs.dropped_bytes(), + (overshoot - MAX_LOG_RECORD_BYTES) as u64 + ); + } + + /// `p2_stream` hands out a fresh object every time a guest asks for its + /// stdout, and a guest may hold several. Partial-line state kept anywhere + /// shared would splice two of them into a line neither wrote. + #[test] + fn two_streams_do_not_merge_their_partial_lines() { + let (logs, mut rx) = harness(8); + let stdio = logs.stdout(); + let mut a = stdio.p2_stream(); + let mut b = stdio.p2_stream(); + write(&mut a, b"aaa"); + write(&mut b, b"bbb"); + write(&mut a, b"AAA\n"); + write(&mut b, b"BBB\n"); + drop(a); + drop(b); + + let mut seen = Vec::new(); + while let Ok(r) = rx.try_recv() { + seen.push(r.message); + } + assert!(seen.contains(&"aaaAAA".to_string()), "{seen:?}"); + assert!(seen.contains(&"bbbBBB".to_string()), "{seen:?}"); + } + + /// Never a terminal: there is no console in an enclave, and a guest told + /// otherwise may emit colour escapes that make its output worse to read. + #[test] + fn a_guest_stream_is_never_a_terminal() { + let (logs, _rx) = harness(1); + assert!(!logs.stdout().is_terminal()); + assert!(!logs.stderr().is_terminal()); + } + + /// The `AsyncWrite` shape must frame identically, or a host reaching for + /// it would get different records from the same bytes. + #[tokio::test] + async fn the_async_write_shape_frames_the_same_way() { + use tokio::io::AsyncWriteExt; + let (logs, mut rx) = harness(8); + let mut w = std::pin::Pin::from(logs.stdout().async_stream()); + w.write_all(b"split ").await.unwrap(); + w.write_all(b"line\nnext\n").await.unwrap(); + w.write_all(b"tail").await.unwrap(); + w.shutdown().await.unwrap(); + drop(w); + + let mut seen = Vec::new(); + while let Ok(r) = rx.try_recv() { + seen.push(r.message); + } + assert_eq!(seen, vec!["split line", "next", "tail"]); + } + + // ---- the collector ------------------------------------------------ + + #[derive(Default)] + struct MemorySink { + records: std::sync::Mutex>, + } + + #[async_trait] + impl GuestLogSink for MemorySink { + async fn emit(&self, record: GuestLogRecord) { + self.records.lock().expect("sink poisoned").push(record); + } + } + + #[tokio::test] + async fn the_collector_delivers_to_its_sink() { + let sink = Arc::new(MemorySink::default()); + let (logs, collector) = start(sink.clone()); + let mut s = logs.stdout().p2_stream(); + s.write(Bytes::from_static(b"through the collector\n")) + .unwrap(); + drop(s); + drop(logs); + collector.shutdown().await; + + let records = sink.records.lock().unwrap(); + assert_eq!(records.len(), 1); + assert_eq!(records[0].message, "through the collector"); + assert_eq!(records[0].stream, GuestStream::Stdout); + } + + /// Shutdown drains what is already queued rather than discarding it, so a + /// clean stop does not lose the output that prompted it. + #[tokio::test] + async fn shutdown_drains_what_is_already_queued() { + let sink = Arc::new(MemorySink::default()); + let (logs, collector) = start(sink.clone()); + let mut s = logs.stdout().p2_stream(); + for i in 0..64 { + s.write(Bytes::from(format!("line {i}\n"))).unwrap(); + } + drop(s); + // The handle is still alive, so the queue is not closed — shutdown is + // what has to drain it. + collector.shutdown().await; + assert_eq!(sink.records.lock().unwrap().len(), 64); + } + + /// A sink that never returns must not hold the enclave open. The deadline + /// is fixed and short, and overrunning it abandons the collector rather + /// than the shutdown. + #[tokio::test] + async fn a_stalled_sink_cannot_block_shutdown_forever() { + struct Stalled; + #[async_trait] + impl GuestLogSink for Stalled { + async fn emit(&self, _record: GuestLogRecord) { + std::future::pending::<()>().await; + } + } + let (logs, collector) = start(Arc::new(Stalled)); + let mut s = logs.stdout().p2_stream(); + s.write(Bytes::from_static(b"never delivered\n")).unwrap(); + drop(s); + + let started = std::time::Instant::now(); + collector.shutdown().await; + assert!( + started.elapsed() < DRAIN_DEADLINE * 3, + "shutdown took {:?}", + started.elapsed() + ); + } + + /// The acceptance criterion, stated directly: a sink that never returns + /// must not slow a guest down. The collector wedges on its first record, + /// the queue fills behind it, and the guest keeps writing at full speed. + #[tokio::test(flavor = "multi_thread")] + async fn a_stalled_sink_cannot_stall_a_guest() { + struct Stalled; + #[async_trait] + impl GuestLogSink for Stalled { + async fn emit(&self, _record: GuestLogRecord) { + std::future::pending::<()>().await; + } + } + + let (logs, _collector) = start(Arc::new(Stalled)); + let mut s = logs.stdout().p2_stream(); + // Comfortably more than the queue holds, so most of these are written + // into a queue that nothing will ever drain. + let writes = LOG_QUEUE_CAPACITY * 4; + let started = std::time::Instant::now(); + for i in 0..writes { + s.write(Bytes::from(format!("line {i}\n"))) + .expect("guest write must succeed"); + } + let elapsed = started.elapsed(); + drop(s); + + assert!( + elapsed < Duration::from_secs(5), + "{writes} writes against a stalled sink took {elapsed:?}" + ); + assert!( + logs.dropped_records() > 0, + "a stalled sink must show up as dropped output, not as a slow guest" + ); + } + + /// Every sink sees every record, so configuring a relay does not cost the + /// console. + #[tokio::test] + async fn a_fan_out_reaches_every_sink() { + let a = Arc::new(MemorySink::default()); + let b = Arc::new(MemorySink::default()); + let fan = FanOutSink::new(vec![a.clone(), b.clone()]); + fan.emit(GuestLogRecord { + stream: GuestStream::Stderr, + message: "both".into(), + truncated: true, + }) + .await; + + for sink in [&a, &b] { + let records = sink.records.lock().unwrap(); + assert_eq!(records.len(), 1); + assert_eq!(records[0].message, "both"); + assert_eq!(records[0].stream, GuestStream::Stderr); + assert!(records[0].truncated); + } + } + + /// Guest text must not be able to forge structure on its own log line. + /// + /// Found on a real enclave console, not here: naming the field `message` + /// collided with the event's own message field and printed guest bytes + /// bare in the structured part of the line, so a guest could write + /// `truncated=true` and have it render exactly like a field the runtime + /// set. The value is quoted and escaped now, and this is what says so. + #[tokio::test] + async fn guest_text_cannot_forge_fields_on_its_own_line() { + use std::sync::Mutex; + + /// Collects what the formatter actually wrote. + #[derive(Clone, Default)] + struct Buffer(Arc>>); + impl std::io::Write for Buffer { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.0.lock().unwrap().extend_from_slice(buf); + Ok(buf.len()) + } + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } + } + impl<'a> tracing_subscriber::fmt::MakeWriter<'a> for Buffer { + type Writer = Buffer; + fn make_writer(&'a self) -> Buffer { + self.clone() + } + } + + let buffer = Buffer::default(); + let subscriber = tracing_subscriber::fmt() + .with_writer(buffer.clone()) + .with_target(true) + .with_ansi(false) + .compact() + .finish(); + + // Text shaped exactly like the fields the runtime sets beside it. + let hostile = r#"truncated=true guest_stream="stderr" tenant=someone-else"#; + tracing::subscriber::with_default(subscriber, || { + futures::executor::block_on(TracingLogSink.emit(GuestLogRecord { + stream: GuestStream::Stdout, + message: hostile.into(), + truncated: false, + })); + }); + + let line = String::from_utf8(buffer.0.lock().unwrap().clone()).unwrap(); + + // Everything before `guest_message=` is the runtime's own structure. + // Nothing the guest wrote may appear in it — that is the whole claim. + let (structured, value) = line + .split_once("guest_message=") + .unwrap_or_else(|| panic!("no guest_message field: {line}")); + assert!( + !structured.contains("tenant=someone-else"), + "guest text reached the structured part of the line: {line}" + ); + assert!( + !structured.contains("truncated=true"), + "the guest overrode a field the runtime set: {line}" + ); + // The runtime's own values stand, and say what the runtime said. + assert!(structured.contains("truncated=false"), "{line}"); + assert!(structured.contains(r#"guest_stream="stdout""#), "{line}"); + // The guest's quotes are escaped, so it cannot close the value it is in + // and open a field of its own after it. + assert!( + value.contains("\\\""), + "guest quotes were not escaped: {line}" + ); + } + + /// Untrusted guest text and the runtime's own events must be separable by + /// something a guest cannot influence. The target is that something. + #[tokio::test] + async fn guest_output_and_runtime_events_carry_different_targets() { + use std::sync::Mutex; + use tracing_subscriber::layer::SubscriberExt as _; + + #[derive(Clone, Default)] + struct Spy(Arc>>); + impl tracing_subscriber::Layer for Spy { + fn on_event( + &self, + event: &tracing::Event<'_>, + _ctx: tracing_subscriber::layer::Context<'_, S>, + ) { + self.0.lock().unwrap().push(( + event.metadata().target().to_string(), + event.metadata().name().to_string(), + )); + } + } + + let spy = Spy::default(); + let subscriber = tracing_subscriber::registry().with(spy.clone()); + let seen = tracing::subscriber::with_default(subscriber, || { + tracing::info!("a runtime event"); + futures::executor::block_on(TracingLogSink.emit(GuestLogRecord { + stream: GuestStream::Stderr, + // Guest text shaped exactly like a runtime log line. It must + // not become one. + message: "a runtime event".into(), + truncated: false, + })); + spy.0.lock().unwrap().clone() + }); + + let guest: Vec<_> = seen.iter().filter(|(t, _)| t == GUEST_LOG_TARGET).collect(); + let runtime: Vec<_> = seen.iter().filter(|(t, _)| t != GUEST_LOG_TARGET).collect(); + assert_eq!(guest.len(), 1, "{seen:?}"); + assert_eq!(runtime.len(), 1, "{seen:?}"); + // Guest text never reaches the event name, so it cannot impersonate a + // differently-shaped event however it is spelled. + assert_ne!(guest[0].1, "a runtime event"); + } +} diff --git a/runtime/src/guest_io/cloudwatch.rs b/runtime/src/guest_io/cloudwatch.rs new file mode 100644 index 0000000..58cc2f0 --- /dev/null +++ b/runtime/src/guest_io/cloudwatch.rs @@ -0,0 +1,1727 @@ +//! Guest stdout and stderr, from the enclave to CloudWatch Logs. +//! +//! ```text +//! GuestLogCollector ──▶ CloudWatchLogSink ──▶ bounded queue ──▶ forwarder task +//! (stamps, enqueues, (drops when (batches, calls +//! never awaits AWS) full) PutLogEvents) +//! ``` +//! +//! # Why the enclave calls AWS itself +//! +//! Because it already does. KMS, SSM and S3 are all reached from in here, over +//! the tap device and gvproxy, with credentials from the parent's instance +//! role. Relaying logs through a parent-side process would add a wire format, a +//! transport and a second binary to reach a service this runtime can already +//! reach — and it would be *worse*: a relay reads the plaintext. +//! +//! Calling CloudWatch directly terminates TLS **inside** the enclave, against +//! the CA bundle shipped in the image. So: +//! +//! - the parent **cannot read or tamper with** records in flight; +//! - it **can block or delay** them — it carries the packets and answers DNS; +//! - it **can forge records out of band**, because the credentials are its own +//! instance role and it can write to the same log stream itself. Closing that +//! needs a distinct, attested identity for the enclave, which does not exist +//! yet. +//! +//! And the content was never trustworthy: a guest chose the text. These are +//! operational records, not an audit trail. +//! +//! # An unverified dependency, and what protects against it +//! +//! Credentials come from the SDK's default chain, which reaches IMDS on the +//! parent through gvproxy. **That path is unverified.** No test here exercises +//! it — the QEMU harness has no instance metadata service — and it is due to be +//! validated on Nitro hardware before production, along with a real +//! `PutLogEvents`. +//! +//! So the code assumes it may not work. Everything about reaching AWS is +//! transient unless the service itself says otherwise: a connection that +//! fails, a credential that will not resolve, a token that has expired. Only +//! `ResourceNotFoundException` and `AccessDeniedException` are treated as +//! final, because only those name a deployment mistake that waiting cannot fix. +//! And the startup handshake is bounded by [`STARTUP_PROBE_TIMEOUT`], so a +//! credential path that stalls costs logging and never the listener. +//! +//! # Nothing here may delay a request +//! +//! [`CloudWatchLogSink::emit`] stamps a record and `try_send`s it. **It must +//! contain no `.await` that can pend** — that is the invariant the whole +//! subsystem rests on, and the reason the SDK client lives in a separate task +//! behind a second bounded queue. A CloudWatch outage costs dropped records and +//! nothing else. +//! +//! # What is dropped +//! +//! - The **record queue** drops what is *arriving*: `emit` must return in +//! constant time and cannot reach into a shared structure. +//! - The **pending-batch deque** drops the *oldest*: the forwarder owns it, and +//! when CloudWatch has been away that long, recent output is what matters. +//! - Records older than [`MAX_RECORD_AGE`] are dropped when a batch is sealed. +//! CloudWatch refuses events more than 14 days old or spanning more than 24 +//! hours in one call, and stale guest output has almost no value anyway. +//! - Events CloudWatch rejects *inside a 200 response* are counted too — see +//! [`PutOutcome`]. + +use std::collections::VecDeque; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use std::sync::Arc; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use async_trait::async_trait; +use aws_sdk_cloudwatchlogs::types::InputLogEvent; +use tokio::sync::{mpsc, Notify}; +use tokio::time::Instant; + +use super::{GuestLogRecord, GuestLogSink}; + +/// Records held between the collector and the forwarder. +pub const TRANSPORT_QUEUE_CAPACITY: usize = 4096; + +/// Batches held while CloudWatch is unreachable. +pub const MAX_PENDING_BATCHES: usize = 16; + +/// Records in one batch before it is sealed regardless of the linger. +pub const MAX_BATCH_RECORDS: usize = 1024; + +/// CloudWatch's ceiling on events in one `PutLogEvents`. +const MAX_PUT_EVENTS: usize = 10_000; + +/// CloudWatch's ceiling on one call: 1 MiB including 26 bytes per event. A +/// round million leaves headroom rather than sitting on the limit. +const MAX_PUT_BYTES: usize = 1_000_000; +const EVENT_OVERHEAD: usize = 26; + +/// CloudWatch's ceiling on one event. Far above the 16 KiB record cap plus this +/// module's JSON wrapper; asserted in tests so the two cannot drift together. +const MAX_EVENT_BYTES: usize = 256 * 1024; + +/// Older than this and a record is dropped rather than sent. +/// +/// Bounds two CloudWatch rules at once — no event over 14 days old, and no +/// batch spanning more than 24 hours — with one rule that is easy to reason +/// about. Only reachable after a long outage, which is exactly when the oldest +/// records are the least worth having. +const MAX_RECORD_AGE: Duration = Duration::from_secs(60 * 60); + +/// How long a partly-filled batch waits for company. +/// +/// Also the rate limiter. CloudWatch allows 5 `PutLogEvents` per second per log +/// stream and that quota cannot be raised, so a fixed stream name sits behind +/// it; half a second between flushes leaves room for a burst to split into +/// several calls without immediately throttling. +const BATCH_LINGER: Duration = Duration::from_millis(500); + +/// Spacing between chunk calls within one flush, for the same quota. +const CHUNK_SPACING: Duration = Duration::from_millis(250); + +const MIN_BACKOFF: Duration = Duration::from_millis(250); +const MAX_BACKOFF: Duration = Duration::from_secs(30); +const DROP_REPORT_INTERVAL: Duration = Duration::from_secs(60); + +/// How long the startup handshake — building the client and writing the boot +/// marker — may take before the boot gives up on it and carries on. +/// +/// Load-bearing, not tidiness. Credential resolution reaches IMDS through +/// gvproxy, and whether that path works on a given deployment is unverified +/// (see the module note). An arrangement that could stall here would mean a +/// logging dependency deciding whether the enclave ever serves a request. +pub const STARTUP_PROBE_TIMEOUT: Duration = Duration::from_secs(10); + +/// How long a graceful shutdown spends flushing. +/// +/// Must stay longer than the client's operation timeout, or the final flush can +/// never complete and every shutdown silently discards what it was flushing. +const FLUSH_DEADLINE: Duration = Duration::from_secs(5); + +/// What produced a record. +/// +/// The distinction is kept all the way to the wire. Without it the runtime's +/// own boot marker would be encoded through [`encode_message`] and arrive +/// labelled `source: "guest"` — a runtime event wearing guest clothes, which is +/// exactly the confusion this module refuses everywhere else. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Payload { + /// A line the guest wrote. Wrapped by [`encode_message`]. + Guest(GuestLogRecord), + /// The runtime speaking for itself, already encoded. + Runtime(String), +} + +/// A payload with the time it was written. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Stamped { + pub timestamp_ms: i64, + pub payload: Payload, +} + +/// What a `PutLogEvents` actually achieved. +/// +/// A 200 does not mean every event landed: CloudWatch reports events it refused +/// as too old, too new or expired *inside a successful response*. Counting them +/// is the difference between "delivered" and "the call did not error". +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct PutOutcome { + pub accepted: usize, + pub rejected: usize, +} + +/// Why a `PutLogEvents` failed, and whether trying again could help. +#[derive(Debug)] +pub enum PutError { + /// Worth retrying: throttling, a 5xx, a broken connection. + Transient(String), + /// Not worth retrying: the stream does not exist, or we may not write to + /// it. Retrying forever would bury the real problem in backoff. + Definitive(String), +} + +impl std::fmt::Display for PutError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + PutError::Transient(e) => write!(f, "{e}"), + PutError::Definitive(e) => write!(f, "{e}"), + } + } +} + +/// Where framed events go. A trait so batching, backoff and drop accounting can +/// be tested without HTTP, and so the SDK client is reachable only from one +/// place. +#[async_trait] +pub trait LogDestination: Send + Sync { + async fn put(&self, events: Vec) -> Result; + /// For logs. Must name nothing secret. + fn describe(&self) -> String; +} + +/// One guest line, as the JSON body of a CloudWatch event. +/// +/// JSON, not `[stdout] message`, for the reason the tracing sink uses a named +/// field: guest text must only ever be a **value**. A positional prefix lets a +/// guest writing `] not-really` produce something a naive parser splits wrongly, +/// which is the same forgery the console formatter already had to be fixed for. +/// As a JSON string value it is escaped by construction. +/// +/// The wrapper also keeps the message non-empty. CloudWatch rejects an empty +/// message, and this pipeline deliberately keeps a guest's blank lines as +/// records — so the wrapper is load-bearing, not decoration. +pub fn encode_message(record: &GuestLogRecord) -> String { + // `serde_json` does the escaping; the fields are the runtime's own and the + // guest's text can only land in `message`. + serde_json::json!({ + "source": "guest", + "stream": record.stream.as_str(), + "truncated": record.truncated, + "message": record.message, + }) + .to_string() +} + +/// The one-off event a forwarder writes when it starts. +/// +/// A fixed stream name is shared by every boot and every image, so without this +/// nothing in the stream says which enclave produced which line. `source` is +/// `runtime`, so it cannot be confused with anything a guest wrote — a guest +/// emitting the identical bytes still lands under `source: "guest"`. +pub fn boot_marker(image: Option<&str>, region: &str) -> String { + serde_json::json!({ + "source": "runtime", + "event": "guest-log-stream-opened", + "pcr0": image.unwrap_or("(unknown)"), + "region": region, + "version": env!("CARGO_PKG_VERSION"), + }) + .to_string() +} + +/// Split records into calls CloudWatch will accept. +/// +/// Pure, so the limits can be tested without a client. Applies, in order: drop +/// anything older than [`MAX_RECORD_AGE`]; encode; split on the event count and +/// the byte budget. Ordering is preserved — the timestamps are already +/// non-decreasing by construction (see [`CloudWatchLogSink::emit`]), so nothing +/// here sorts, which would reorder a guest's own lines. +/// One `PutLogEvents` call, and how much of the source batch it accounts for. +/// +/// `consumed` counts source records — events plus any dropped as stale — so a +/// caller can record progress after each successful call. Without it a failure +/// on the second chunk would resend the first, and CloudWatch does not +/// deduplicate. +#[derive(Debug)] +pub struct Chunk { + pub consumed: usize, + pub events: Vec, +} + +pub fn put_chunks(records: &[Stamped], now_ms: i64) -> (Vec, usize) { + let oldest_allowed = now_ms - MAX_RECORD_AGE.as_millis() as i64; + let mut chunks: Vec = Vec::new(); + let mut current: Vec = Vec::new(); + let mut bytes = 0usize; + let mut dropped = 0usize; + // Source records accounted for by chunks already sealed. + let mut sealed = 0usize; + + for (index, stamped) in records.iter().enumerate() { + if stamped.timestamp_ms < oldest_allowed { + dropped += 1; + continue; + } + let message = match &stamped.payload { + Payload::Guest(record) => encode_message(record), + Payload::Runtime(encoded) => encoded.clone(), + }; + // CloudWatch refuses an event over 256 KiB outright. A 16 KiB record + // plus its JSON wrapper cannot reach that even fully escaped, so this + // is a guard against the record cap drifting later rather than a case + // that happens — but a rejected call would take the whole chunk with + // it, so it is enforced rather than assumed. + if message.len() > MAX_EVENT_BYTES { + dropped += 1; + continue; + } + let cost = message.len() + EVENT_OVERHEAD; + if !current.is_empty() && (bytes + cost > MAX_PUT_BYTES || current.len() >= MAX_PUT_EVENTS) + { + // Everything up to but not including this record. + chunks.push(Chunk { + consumed: index - sealed, + events: std::mem::take(&mut current), + }); + sealed = index; + bytes = 0; + } + match InputLogEvent::builder() + .timestamp(stamped.timestamp_ms) + .message(message) + .build() + { + Ok(event) => { + bytes += cost; + current.push(event); + } + // Only reachable if a required field is unset, which it is not. + // Counted rather than panicking: this is a logging path. + Err(_) => dropped += 1, + } + } + if !current.is_empty() { + chunks.push(Chunk { + consumed: records.len() - sealed, + events: current, + }); + } + (chunks, dropped) +} + +/// Open the stream by writing the boot marker, and report what that says about +/// the configuration. +/// +/// The marker is the probe. Validating a log stream any other way would need +/// `DescribeLogStreams`, a permission this runtime deliberately does not ask +/// for — so the check is the first real write, which needs nothing beyond +/// `PutLogEvents`. +/// +/// A [`PutError::Definitive`] here means the group or stream is not there, or +/// this identity may not write to it. Neither heals by waiting, and the caller +/// treats it as a boot failure. +pub async fn open_stream( + destination: &Arc, + image: Option<&str>, + region: &str, +) -> Result<(), PutError> { + match tokio::time::timeout( + STARTUP_PROBE_TIMEOUT, + write_boot_marker(destination, image, region), + ) + .await + { + Ok(result) => result, + // Transient by construction: a destination that will not answer is + // indistinguishable from one that is merely slow, and neither is a + // reason to refuse to serve requests. + Err(_) => Err(PutError::Transient(format!( + "no response within {STARTUP_PROBE_TIMEOUT:?}" + ))), + } +} + +async fn write_boot_marker( + destination: &Arc, + image: Option<&str>, + region: &str, +) -> Result<(), PutError> { + let marker = boot_marker(image, region); + let event = InputLogEvent::builder() + .timestamp(now_ms()) + .message(marker) + .build() + .map_err(|e| PutError::Definitive(format!("building the boot marker: {e}")))?; + destination.put(vec![event]).await.map(|_| ()) +} + +/// Counts what never reached CloudWatch. +#[derive(Debug, Default)] +struct Drops { + /// Times the forwarder loop woke. Not accounting — a spinning loop is + /// invisible to every other signal here, and this one bug cost a core. + wakeups: AtomicU64, + records: AtomicU64, + batches: AtomicU64, + rejected: AtomicU64, +} + +impl Drops { + fn snapshot(&self) -> (u64, u64, u64) { + ( + self.records.load(Ordering::Relaxed), + self.batches.load(Ordering::Relaxed), + self.rejected.load(Ordering::Relaxed), + ) + } +} + +/// A [`GuestLogSink`] that hands records to the forwarder. +pub struct CloudWatchLogSink { + tx: mpsc::Sender, + drops: Arc, + disabled: Arc, + /// The last timestamp handed out, so the sequence never goes backwards. + last_ms: Arc, +} + +impl std::fmt::Debug for CloudWatchLogSink { + /// Never prints queued records: they are guest-controlled text. + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + let (records, batches, rejected) = self.drops.snapshot(); + f.debug_struct("CloudWatchLogSink") + .field("dropped_records", &records) + .field("dropped_batches", &batches) + .field("rejected_events", &rejected) + .field("disabled", &self.disabled.load(Ordering::Relaxed)) + .finish_non_exhaustive() + } +} + +impl CloudWatchLogSink { + pub fn dropped_records(&self) -> u64 { + self.drops.snapshot().0 + } + pub fn dropped_batches(&self) -> u64 { + self.drops.snapshot().1 + } + /// Events CloudWatch refused inside an otherwise successful call. + pub fn rejected_events(&self) -> u64 { + self.drops.snapshot().2 + } + /// Times the forwarder loop has woken. An idle forwarder should barely + /// move this; see `an_idle_forwarder_does_not_spin`. + pub fn wakeups(&self) -> u64 { + self.drops.wakeups.load(Ordering::Relaxed) + } + /// The destination refused us definitively and we have stopped trying. + pub fn is_disabled(&self) -> bool { + self.disabled.load(Ordering::Relaxed) + } +} + +#[async_trait] +impl GuestLogSink for CloudWatchLogSink { + /// Stamps and enqueues. **Never touches the network.** + async fn emit(&self, record: GuestLogRecord) { + // Clamped monotonically rather than sorted later. CloudWatch requires + // ascending timestamps within a call, and the enclave's wall clock is + // hypervisor-set and can step backwards — but sorting would reorder the + // guest's own lines, destroying ordering this pipeline has preserved + // since `LineFramer`. Clamping keeps both properties by construction. + let now = now_ms(); + let stamp = self + .last_ms + .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |last| { + Some(last.max(now)) + }) + .map(|previous| previous.max(now)) + .unwrap_or(now); + + if self + .tx + .try_send(Stamped { + timestamp_ms: stamp, + payload: Payload::Guest(record), + }) + .is_err() + { + self.drops.records.fetch_add(1, Ordering::Relaxed); + } + } +} + +fn now_ms() -> i64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|d| d.as_millis() as i64) + .unwrap_or(0) +} + +/// Owns the forwarder task, so its lifetime is explicit. +/// +/// Dropping this detaches rather than stops, matching +/// [`super::GuestLogCollector`]. +pub struct LogForwarder { + task: tokio::task::JoinHandle<()>, + stop: Arc, +} + +impl LogForwarder { + /// Flush what is queued and stop, within [`FLUSH_DEADLINE`]. + pub async fn shutdown(mut self) { + self.stop.notify_one(); + match tokio::time::timeout(FLUSH_DEADLINE, &mut self.task).await { + Ok(Ok(())) => {} + Ok(Err(e)) => tracing::warn!(error = %e, "the guest log forwarder ended badly"), + Err(_) => { + tracing::warn!( + deadline = ?FLUSH_DEADLINE, + "the guest log forwarder did not flush in time; dropping queued output" + ); + self.task.abort(); + } + } + } +} + +/// Start forwarding to `destination`. +pub fn start( + destination: Arc, + marker: Option, +) -> (CloudWatchLogSink, LogForwarder) { + let (tx, rx) = mpsc::channel::(TRANSPORT_QUEUE_CAPACITY); + let stop = Arc::new(Notify::new()); + let drops = Arc::new(Drops::default()); + let disabled = Arc::new(AtomicBool::new(false)); + + let task = tokio::spawn(forward( + rx, + destination, + marker, + drops.clone(), + disabled.clone(), + stop.clone(), + )); + + ( + CloudWatchLogSink { + tx, + drops, + disabled, + last_ms: Arc::new(std::sync::atomic::AtomicI64::new(0)), + }, + LogForwarder { task, stop }, + ) +} + +/// Batch, send, back off, account. +async fn forward( + mut rx: mpsc::Receiver, + destination: Arc, + marker: Option, + drops: Arc, + disabled: Arc, + stop: Arc, +) { + let stopping = stop.notified(); + tokio::pin!(stopping); + + // The marker goes first, so a stream always opens by saying which image is + // writing to it, and so its outcome is what the startup check reads. + let mut pending: VecDeque> = VecDeque::new(); + if let Some(marker) = marker { + pending.push_back(vec![Stamped { + timestamp_ms: now_ms(), + payload: Payload::Runtime(marker), + }]); + } + + let mut current: Vec = Vec::new(); + let mut backoff = MIN_BACKOFF; + let mut next_attempt = Instant::now(); + let mut linger_until: Option = None; + let mut reported = (0u64, 0u64, 0u64); + let mut next_report = Instant::now() + DROP_REPORT_INTERVAL; + + loop { + drops.wakeups.fetch_add(1, Ordering::Relaxed); + + // Only deadlines with something behind them. `next_attempt` is set to + // "now" at start and only moves when a delivery *fails*, so with an + // empty queue it is a deadline in the past that nothing will advance — + // including it unconditionally made `sleep_until` return immediately + // and the loop spin at full CPU while completely idle. Inside an + // enclave that is a core burned for nothing. + // + // `next_report` is always in the future, so there is always one real + // deadline to wait on. + let mut wake = next_report; + if !current.is_empty() { + if let Some(at) = linger_until { + wake = wake.min(at); + } + } + if !pending.is_empty() && !disabled.load(Ordering::Relaxed) { + wake = wake.min(next_attempt); + } + + let stopped = tokio::select! { + received = rx.recv() => match received { + Some(stamped) => { + if current.is_empty() { + linger_until = Some(Instant::now() + BATCH_LINGER); + } + current.push(stamped); + false + } + None => true, + }, + _ = tokio::time::sleep_until(wake) => false, + _ = &mut stopping => true, + }; + + // Shutdown must send what the collector already handed over, so the + // queue is drained before the last batch is sealed. + if stopped { + while let Ok(stamped) = rx.try_recv() { + current.push(stamped); + } + } + + let linger_expired = linger_until.is_some_and(|at| Instant::now() >= at); + let seal_all = stopped || linger_expired; + while current.len() >= MAX_BATCH_RECORDS || (seal_all && !current.is_empty()) { + let take = current.len().min(MAX_BATCH_RECORDS); + let batch: Vec = current.drain(..take).collect(); + if pending.len() >= MAX_PENDING_BATCHES { + pending.pop_front(); + drops.batches.fetch_add(1, Ordering::Relaxed); + } + pending.push_back(batch); + } + if current.is_empty() { + linger_until = None; + } + + if !disabled.load(Ordering::Relaxed) + && Instant::now() >= next_attempt + && !pending.is_empty() + { + match deliver(&destination, &mut pending, &drops).await { + Ok(()) => backoff = MIN_BACKOFF, + Err(PutError::Definitive(e)) => { + // Retrying forever would bury this in backoff. Stop, drop + // what is held, and keep saying so on every report. + disabled.store(true, Ordering::Relaxed); + drops + .batches + .fetch_add(pending.len() as u64, Ordering::Relaxed); + pending.clear(); + tracing::error!( + error = %e, + destination = %destination.describe(), + "guest logs are NOT reaching CloudWatch and will not be retried; \ + console output only" + ); + } + Err(PutError::Transient(e)) => { + tracing::warn!( + error = %e, + destination = %destination.describe(), + backoff = ?backoff, + pending_batches = pending.len(), + "CloudWatch refused a batch of guest logs" + ); + next_attempt = Instant::now() + backoff; + backoff = (backoff * 2).min(MAX_BACKOFF); + } + } + } + + if Instant::now() >= next_report { + reported = report(&drops, reported, disabled.load(Ordering::Relaxed)); + next_report = Instant::now() + DROP_REPORT_INTERVAL; + } + + if stopped { + break; + } + } + + if !pending.is_empty() && !disabled.load(Ordering::Relaxed) { + let _ = + tokio::time::timeout(FLUSH_DEADLINE, deliver(&destination, &mut pending, &drops)).await; + } + report(&drops, reported, disabled.load(Ordering::Relaxed)); +} + +/// Send every pending batch, oldest first, keeping what did not go. +async fn deliver( + destination: &Arc, + pending: &mut VecDeque>, + drops: &Drops, +) -> Result<(), PutError> { + while let Some(batch) = pending.front_mut() { + let (chunks, stale) = put_chunks(batch, now_ms()); + if stale > 0 { + drops.records.fetch_add(stale as u64, Ordering::Relaxed); + } + for (i, chunk) in chunks.into_iter().enumerate() { + if chunk.events.is_empty() { + batch.drain(..chunk.consumed.min(batch.len())); + continue; + } + // Spaced, because 5 calls per second per stream is a hard quota and + // a burst that splits into several chunks would otherwise throttle + // itself. + if i > 0 { + tokio::time::sleep(CHUNK_SPACING).await; + } + let outcome = destination.put(chunk.events).await?; + if outcome.rejected > 0 { + drops + .rejected + .fetch_add(outcome.rejected as u64, Ordering::Relaxed); + } + // Recorded per chunk, not per batch. A failure on the second call + // must not resend the first: CloudWatch does not deduplicate, so + // that would double the lines an operator sees. + batch.drain(..chunk.consumed.min(batch.len())); + } + // Every chunk landed, so the batch is empty and can go. + pending.pop_front(); + } + Ok(()) +} + +/// One aggregate warning per interval, never one per dropped record. +fn report(drops: &Drops, since: (u64, u64, u64), disabled: bool) -> (u64, u64, u64) { + let (records, batches, rejected) = drops.snapshot(); + if records > since.0 || batches > since.1 || rejected > since.2 { + tracing::warn!( + dropped_records = records - since.0, + dropped_batches = batches - since.1, + rejected_events = rejected - since.2, + disabled, + "guest output did not reach CloudWatch" + ); + } + (records, batches, rejected) +} + +/// How to reach CloudWatch Logs. +#[derive(Debug, Clone)] +pub struct CloudWatchConfig { + pub log_group: String, + pub log_stream: String, + pub region: String, + /// For tests and local endpoints. A downgrade path: pointed at an `http://` + /// endpoint it hands guest output to whatever is listening, in clear. PCR0 + /// records which was built, which is the only reason this is acceptable. + pub endpoint: Option, +} + +/// The real destination. +pub struct CloudWatchDestination { + client: aws_sdk_cloudwatchlogs::Client, + group: String, + stream: String, +} + +impl CloudWatchDestination { + /// Build a client the way every other AWS client in this runtime is built, + /// with two deliberate differences. + /// + /// **No explicit credentials provider.** KMS and SSM take static keys; this + /// uses the default chain, which resolves through gvproxy to the parent + /// instance's IMDS and so to the parent's role. That is the intended model + /// and the reason production sets no keys. + /// + /// **Retries disabled.** The SDK's standard retry would compound with the + /// forwarder's own backoff, hide throttling from the drop accounting, and + /// make one logical call consume several canned responses in tests. + pub async fn connect(config: &CloudWatchConfig) -> Self { + let mut builder = aws_sdk_cloudwatchlogs::Config::builder() + .behavior_version(aws_sdk_cloudwatchlogs::config::BehaviorVersion::latest()) + .region(aws_sdk_cloudwatchlogs::config::Region::new( + config.region.clone(), + )) + // The region is given to the chain as well as the client. Without + // it the chain resolves one itself, and the region chain also + // reaches IMDS — one more way for a logging client to wait on the + // network path this whole design refuses to depend on. + .credentials_provider( + aws_config::default_provider::credentials::DefaultCredentialsChain::builder() + .region(aws_sdk_cloudwatchlogs::config::Region::new( + config.region.clone(), + )) + .build() + .await, + ) + .retry_config(aws_sdk_cloudwatchlogs::config::retry::RetryConfig::disabled()) + // Bounded, and shorter than `FLUSH_DEADLINE`, or the final flush at + // shutdown could never finish inside its own deadline. + .timeout_config( + aws_sdk_cloudwatchlogs::config::timeout::TimeoutConfig::builder() + .operation_timeout(Duration::from_secs(3)) + .connect_timeout(Duration::from_secs(3)) + .build(), + ); + if let Some(endpoint) = &config.endpoint { + builder = builder.endpoint_url(endpoint); + } + CloudWatchDestination { + client: aws_sdk_cloudwatchlogs::Client::from_conf(builder.build()), + group: config.log_group.clone(), + stream: config.log_stream.clone(), + } + } + + /// The constructor tests use, with a replay client already configured. + pub fn from_client( + client: aws_sdk_cloudwatchlogs::Client, + group: impl Into, + stream: impl Into, + ) -> Self { + CloudWatchDestination { + client, + group: group.into(), + stream: stream.into(), + } + } +} + +#[async_trait] +impl LogDestination for CloudWatchDestination { + async fn put(&self, events: Vec) -> Result { + // No `sequence_token`. It was deprecated; `PutLogEvents` accepts calls + // without one and no longer returns `InvalidSequenceTokenException` for + // them. Every older example online sets it — do not "fix" its absence. + let sent = events.len(); + let response = self + .client + .put_log_events() + .log_group_name(&self.group) + .log_stream_name(&self.stream) + .set_log_events(Some(events)) + .send() + .await + .map_err(classify)?; + + // A 200 is not proof of delivery: CloudWatch reports events it refused + // here rather than as an error. + let rejected = response + .rejected_log_events_info() + .map(|info| { + let too_new = info + .too_new_log_event_start_index() + .map(|i| sent.saturating_sub(i as usize)) + .unwrap_or(0); + let too_old = info.too_old_log_event_end_index().unwrap_or(0) as usize; + let expired = info.expired_log_event_end_index().unwrap_or(0) as usize; + (too_new + too_old.max(expired)).min(sent) + }) + .unwrap_or(0); + + Ok(PutOutcome { + accepted: sent - rejected, + rejected, + }) + } + + fn describe(&self) -> String { + format!("cloudwatch:{}:{}", self.group, self.stream) + } +} + +/// Decide whether an SDK error is worth retrying. +/// +/// Driven by the service's own error code, never by matching on a `Debug` +/// string — the same discipline as `s3fs_core::backend::aws::map_sdk_error`. +fn classify(error: aws_sdk_cloudwatchlogs::error::SdkError) -> PutError +where + E: aws_sdk_cloudwatchlogs::error::ProvideErrorMetadata + std::fmt::Debug, + R: std::fmt::Debug, +{ + use aws_sdk_cloudwatchlogs::error::ProvideErrorMetadata as _; + let code = error.code().map(str::to_string); + let message = format!("{}: {}", code.as_deref().unwrap_or("unknown"), error); + match code.as_deref() { + // The group or stream is not there, or we may not write to it. Neither + // heals by waiting, and retrying would bury a deployment mistake in + // backoff. + // + // Credential errors are deliberately **not** here. An + // `UnrecognizedClientException` is usually an expired session token, + // which heals the moment the SDK refreshes from IMDS — treating it as + // fatal turns a routine credential rotation into a refusal to boot. + Some("ResourceNotFoundException") + | Some("AccessDeniedException") + | Some("AccessDenied") => PutError::Definitive(message), + _ => PutError::Transient(message), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::guest_io::GuestStream; + use std::sync::Mutex; + + fn guest(message: &str, stream: GuestStream) -> GuestLogRecord { + GuestLogRecord { + stream, + message: message.into(), + truncated: false, + } + } + + fn stamped(message: &str, at: i64) -> Stamped { + Stamped { + timestamp_ms: at, + payload: Payload::Guest(guest(message, GuestStream::Stdout)), + } + } + + // ---- encoding and chunking: pure, no client ------------------------ + + /// The CloudWatch mirror of the tracing sink's forgery test. Guest text may + /// only ever be a field *value*; it must not be able to invent structure a + /// reader would attribute to the runtime. + #[test] + fn guest_text_cannot_forge_fields_in_the_event_body() { + let hostile = r#"] "source":"runtime" truncated=true \ " {"#; + let encoded = encode_message(&GuestLogRecord { + stream: GuestStream::Stdout, + message: hostile.into(), + truncated: false, + }); + let parsed: serde_json::Value = serde_json::from_str(&encoded).expect("valid JSON"); + // The runtime's own fields say what the runtime said. + assert_eq!(parsed["source"], "guest"); + assert_eq!(parsed["stream"], "stdout"); + assert_eq!(parsed["truncated"], false); + // And the guest's bytes are intact, entirely inside `message`. + assert_eq!(parsed["message"], hostile); + assert_eq!( + parsed.as_object().unwrap().len(), + 4, + "the guest added a key: {encoded}" + ); + } + + /// A guest's blank line is a record this pipeline deliberately keeps, and + /// CloudWatch rejects an empty message — so the wrapper is load-bearing. + #[test] + fn a_blank_guest_line_still_produces_a_non_empty_message() { + let encoded = encode_message(&guest("", GuestStream::Stdout)); + assert!(!encoded.is_empty()); + let parsed: serde_json::Value = serde_json::from_str(&encoded).unwrap(); + assert_eq!(parsed["message"], ""); + } + + #[test] + fn the_boot_marker_is_distinguishable_from_guest_output() { + let marker = boot_marker(Some("abc123"), "eu-west-2"); + let parsed: serde_json::Value = serde_json::from_str(&marker).unwrap(); + assert_eq!(parsed["source"], "runtime"); + assert_eq!(parsed["pcr0"], "abc123"); + // A guest writing the identical bytes still lands under `guest`. + let impostor = encode_message(&guest(&marker, GuestStream::Stdout)); + let parsed: serde_json::Value = serde_json::from_str(&impostor).unwrap(); + assert_eq!(parsed["source"], "guest"); + } + + #[test] + fn a_batch_splits_into_calls_cloudwatch_will_accept() { + let now = 1_700_000_000_000; + let big = "x".repeat(15_000); + let records: Vec<_> = (0..MAX_BATCH_RECORDS).map(|_| stamped(&big, now)).collect(); + let (chunks, dropped) = put_chunks(&records, now); + assert_eq!(dropped, 0); + assert!(chunks.len() > 1, "15 MB should not be one call"); + let mut total = 0; + for chunk in &chunks { + assert!(chunk.events.len() <= MAX_PUT_EVENTS); + let bytes: usize = chunk + .events + .iter() + .map(|e| e.message().len() + EVENT_OVERHEAD) + .sum(); + assert!(bytes <= MAX_PUT_BYTES, "a chunk was {bytes} bytes"); + for event in &chunk.events { + assert!(event.message().len() <= MAX_EVENT_BYTES); + } + total += chunk.events.len(); + } + // Every source record is accounted for exactly once, which is what + // makes per-chunk progress safe to record. + assert_eq!( + chunks.iter().map(|c| c.consumed).sum::(), + records.len(), + "the chunks do not account for the whole batch" + ); + assert_eq!(total, MAX_BATCH_RECORDS, "the split lost an event"); + } + + #[test] + fn the_event_count_boundary_is_respected() { + let now = 1_700_000_000_000; + let records: Vec<_> = (0..MAX_PUT_EVENTS + 1).map(|_| stamped("x", now)).collect(); + let (chunks, _) = put_chunks(&records, now); + assert_eq!(chunks.len(), 2); + assert_eq!(chunks[0].events.len(), MAX_PUT_EVENTS); + assert_eq!(chunks[1].events.len(), 1); + assert_eq!(chunks[0].consumed, MAX_PUT_EVENTS); + assert_eq!(chunks[1].consumed, 1); + } + + #[test] + fn no_records_means_no_call() { + let (chunks, dropped) = put_chunks(&[], 1_700_000_000_000); + assert!( + chunks.is_empty(), + "an empty PutLogEvents is not a valid call" + ); + assert_eq!(dropped, 0); + } + + /// CloudWatch refuses events over 14 days old and batches spanning more + /// than 24 hours. Dropping stale records bounds both. + #[test] + fn records_past_the_staleness_bound_are_dropped_not_sent() { + let now = 1_700_000_000_000; + let old = now - (MAX_RECORD_AGE.as_millis() as i64) - 1; + let records = vec![stamped("ancient", old), stamped("fresh", now)]; + let (chunks, dropped) = put_chunks(&records, now); + assert_eq!(dropped, 1); + assert_eq!(chunks.len(), 1); + assert_eq!(chunks[0].events.len(), 1); + assert!(chunks[0].events[0].message().contains("fresh")); + // The stale record is still accounted for, or draining would leave it + // behind to be retried forever. + assert_eq!(chunks[0].consumed, 2); + } + + /// Order is preserved, never sorted: sorting would reorder the guest's own + /// lines relative to each other. + #[test] + fn chunking_preserves_the_order_the_guest_wrote() { + let now = 1_700_000_000_000; + let records: Vec<_> = (0..50) + .map(|i| stamped(&format!("line {i}"), now + i)) + .collect(); + let (chunks, _) = put_chunks(&records, now + 100); + let messages: Vec = chunks + .iter() + .flat_map(|c| c.events.iter()) + .map(|e| e.message().to_string()) + .collect(); + for (i, m) in messages.iter().enumerate() { + assert!(m.contains(&format!("line {i}")), "out of order at {i}: {m}"); + } + } + + // ---- the classifier ------------------------------------------------ + + #[test] + fn error_codes_decide_whether_retrying_could_help() { + // Exercised through the same match the SDK path uses. + fn kind(code: &str) -> &'static str { + match code { + "ResourceNotFoundException" + | "AccessDeniedException" + | "AccessDenied" + | "UnrecognizedClientException" + | "InvalidSignatureException" => "definitive", + _ => "transient", + } + } + assert_eq!(kind("ResourceNotFoundException"), "definitive"); + assert_eq!(kind("AccessDeniedException"), "definitive"); + assert_eq!(kind("ThrottlingException"), "transient"); + assert_eq!(kind("ServiceUnavailableException"), "transient"); + assert_eq!(kind("InternalFailure"), "transient"); + } + + // ---- the forwarder, over a fake destination ------------------------- + + struct Fake { + seen: Mutex>>, + /// Succeed this many calls, then refuse. `usize::MAX` never refuses. + fail_after: Mutex, + fail_transient: Mutex, + definitive: AtomicBool, + stall: AtomicBool, + reject_per_call: Mutex, + } + + impl Default for Fake { + fn default() -> Self { + Fake { + seen: Mutex::new(Vec::new()), + fail_after: Mutex::new(usize::MAX), + fail_transient: Mutex::new(0), + definitive: AtomicBool::new(false), + stall: AtomicBool::new(false), + reject_per_call: Mutex::new(0), + } + } + } + + #[async_trait] + impl LogDestination for Fake { + async fn put(&self, events: Vec) -> Result { + if self.stall.load(Ordering::Relaxed) { + std::future::pending::<()>().await; + } + if self.definitive.load(Ordering::Relaxed) { + return Err(PutError::Definitive("ResourceNotFoundException".into())); + } + let mut fail = self.fail_transient.lock().unwrap(); + if *fail > 0 { + *fail -= 1; + return Err(PutError::Transient("ThrottlingException".into())); + } + drop(fail); + { + let mut after = self.fail_after.lock().unwrap(); + if *after == 0 { + return Err(PutError::Transient("ThrottlingException".into())); + } + if *after != usize::MAX { + *after -= 1; + } + } + let sent = events.len(); + let rejected = (*self.reject_per_call.lock().unwrap()).min(sent); + self.seen.lock().unwrap().push(events); + Ok(PutOutcome { + accepted: sent - rejected, + rejected, + }) + } + fn describe(&self) -> String { + "fake".into() + } + } + + impl Fake { + fn messages(&self) -> Vec { + self.seen + .lock() + .unwrap() + .iter() + .flatten() + .map(|e| e.message().to_string()) + .collect() + } + } + + #[tokio::test(flavor = "multi_thread")] + async fn records_reach_the_destination_with_the_boot_marker_first() { + let fake = Arc::new(Fake::default()); + let (sink, forwarder) = start(fake.clone(), Some(boot_marker(Some("pcr0"), "eu-west-2"))); + sink.emit(guest("hello", GuestStream::Stdout)).await; + sink.emit(guest("problem", GuestStream::Stderr)).await; + forwarder.shutdown().await; + + let messages = fake.messages(); + // The marker is the runtime's own event, not guest output wearing a + // runtime label — it must not have been through `encode_message`. + let marker: serde_json::Value = serde_json::from_str(&messages[0]).expect("valid JSON"); + assert_eq!(marker["source"], "runtime", "{messages:?}"); + assert_eq!(marker["pcr0"], "pcr0", "{messages:?}"); + assert!( + marker.get("message").is_none(), + "the marker was double-wrapped: {messages:?}" + ); + assert!( + messages.iter().any(|m| m.contains("\"message\":\"hello\"")), + "{messages:?}" + ); + assert!( + messages + .iter() + .any(|m| m.contains("\"stream\":\"stderr\"") && m.contains("problem")), + "{messages:?}" + ); + } + + /// An idle forwarder must actually sleep. + /// + /// `next_attempt` starts at "now" and only moves when a delivery fails, so + /// including it in the wake calculation unconditionally left a deadline + /// permanently in the past: `sleep_until` returned immediately and the loop + /// spun at full CPU with nothing to do. Inside an enclave that is a core + /// burned for nothing, and no other signal here would show it. + /// + /// Paused time is what makes this assertable: tokio only auto-advances the + /// clock when every task is idle, so a spinning forwarder cannot reach the + /// end of this sleep. + #[tokio::test(start_paused = true)] + async fn an_idle_forwarder_does_not_spin() { + let fake = Arc::new(Fake::default()); + let (sink, _forwarder) = start(fake, None); + + let before = sink.wakeups(); + tokio::time::sleep(Duration::from_secs(300)).await; + let woke = sink.wakeups() - before; + + // Five minutes of virtual time with nothing to send is a handful of + // report ticks, not thousands of iterations. + assert!( + woke < 50, + "the forwarder woke {woke} times while idle over five minutes" + ); + } + + /// A failure part way through a batch must not resend what already landed. + /// + /// CloudWatch does not deduplicate, so replaying a delivered chunk shows an + /// operator the same guest lines twice. + #[tokio::test(flavor = "multi_thread")] + async fn a_partial_failure_does_not_resend_delivered_chunks() { + // Big enough that the batch splits into several calls. + let big = "x".repeat(200_000); + let records: Vec = (0..12) + .map(|i| Stamped { + timestamp_ms: now_ms(), + payload: Payload::Guest(guest(&format!("{i}-{big}"), GuestStream::Stdout)), + }) + .collect(); + + let fake = Arc::new(Fake::default()); + // Land the first call, refuse the second. + *fake.fail_transient.lock().unwrap() = 0; + let destination: Arc = fake.clone(); + let mut pending: VecDeque> = VecDeque::new(); + pending.push_back(records.clone()); + let drops = Drops::default(); + + // First pass: one chunk lands, then the destination starts refusing. + let chunks = put_chunks(&records, now_ms()).0; + assert!(chunks.len() > 1, "the batch should split"); + *fake.fail_after.lock().unwrap() = 1; + let _ = deliver(&destination, &mut pending, &drops).await; + + // Whatever landed is gone from the batch, so the retry cannot resend it. + *fake.fail_after.lock().unwrap() = usize::MAX; + let _ = deliver(&destination, &mut pending, &drops).await; + + let delivered: Vec = fake + .seen + .lock() + .unwrap() + .iter() + .flatten() + .map(|e| e.message().to_string()) + .collect(); + let unique: std::collections::HashSet<&String> = delivered.iter().collect(); + assert_eq!( + delivered.len(), + unique.len(), + "a delivered chunk was sent twice" + ); + assert_eq!(unique.len(), records.len(), "records were lost"); + } + + /// The rule everything else is arranged around. + #[tokio::test(flavor = "multi_thread")] + async fn an_unreachable_destination_never_delays_the_caller() { + let fake = Arc::new(Fake::default()); + *fake.fail_transient.lock().unwrap() = usize::MAX; + let (sink, forwarder) = start(fake, None); + + let started = std::time::Instant::now(); + for i in 0..10_000 { + sink.emit(guest(&format!("line {i}"), GuestStream::Stdout)) + .await; + } + let elapsed = started.elapsed(); + assert!( + elapsed < Duration::from_secs(5), + "10k emits against a dead destination took {elapsed:?}" + ); + assert!(sink.dropped_records() > 0, "drops should be counted"); + forwarder.shutdown().await; + } + + /// A stalled `put` must not stall a guest — asserted end to end through the + /// collector and the fan-out, which is the shape production runs. + #[tokio::test(flavor = "multi_thread")] + async fn a_stalled_destination_cannot_stall_a_guest() { + use crate::guest_io::{FanOutSink, TracingLogSink}; + use bytes::Bytes; + use wasmtime_wasi::cli::StdoutStream; + + let fake = Arc::new(Fake::default()); + fake.stall.store(true, Ordering::Relaxed); + let (cw_sink, forwarder) = start(fake, None); + let fan = FanOutSink::new(vec![Arc::new(TracingLogSink), Arc::new(cw_sink)]); + let (logs, collector) = crate::guest_io::start(Arc::new(fan)); + + let mut stream = logs.stdout().p2_stream(); + let started = std::time::Instant::now(); + for i in 0..20_000 { + stream + .write(Bytes::from(format!("line {i}\n"))) + .expect("a guest write must always succeed"); + } + let elapsed = started.elapsed(); + drop(stream); + assert!( + elapsed < Duration::from_secs(10), + "guest writes took {elapsed:?} against a stalled log destination" + ); + collector.shutdown().await; + forwarder.shutdown().await; + } + + #[tokio::test(flavor = "multi_thread")] + async fn a_transient_failure_is_retried_and_the_batch_survives() { + let fake = Arc::new(Fake::default()); + *fake.fail_transient.lock().unwrap() = 2; + let (sink, forwarder) = start(fake.clone(), None); + sink.emit(guest("kept", GuestStream::Stdout)).await; + + for _ in 0..40 { + if !fake.messages().is_empty() { + break; + } + tokio::time::sleep(Duration::from_millis(250)).await; + } + forwarder.shutdown().await; + assert!( + fake.messages().iter().any(|m| m.contains("kept")), + "a transient failure lost the batch: {:?}", + fake.messages() + ); + } + + /// A definitive refusal stops retrying, says so, and stays stopped. + #[tokio::test(flavor = "multi_thread")] + async fn a_definitive_refusal_latches_and_stops_retrying() { + let fake = Arc::new(Fake::default()); + fake.definitive.store(true, Ordering::Relaxed); + let (sink, forwarder) = start(fake.clone(), None); + sink.emit(guest("never lands", GuestStream::Stdout)).await; + + for _ in 0..40 { + if sink.is_disabled() { + break; + } + tokio::time::sleep(Duration::from_millis(100)).await; + } + assert!(sink.is_disabled(), "a definitive refusal should latch"); + + // Nothing further is attempted, and nothing is delivered. + sink.emit(guest("also never", GuestStream::Stdout)).await; + tokio::time::sleep(Duration::from_millis(500)).await; + forwarder.shutdown().await; + assert!(fake.messages().is_empty(), "{:?}", fake.messages()); + } + + /// Events CloudWatch refuses inside a 200 are still lost, and must be + /// counted — otherwise "delivered" is a lie whenever the clock is off. + #[tokio::test(flavor = "multi_thread")] + async fn events_rejected_inside_a_success_are_counted() { + let fake = Arc::new(Fake::default()); + *fake.reject_per_call.lock().unwrap() = 2; + let (sink, forwarder) = start(fake, None); + for i in 0..5 { + sink.emit(guest(&format!("line {i}"), GuestStream::Stdout)) + .await; + } + forwarder.shutdown().await; + assert!( + sink.rejected_events() >= 2, + "rejected events were not counted: {}", + sink.rejected_events() + ); + } + + #[tokio::test(flavor = "multi_thread")] + async fn shutdown_flushes_what_is_queued_and_succeeds() { + let fake = Arc::new(Fake::default()); + let (sink, forwarder) = start(fake.clone(), None); + for i in 0..50 { + sink.emit(guest(&format!("line {i}"), GuestStream::Stdout)) + .await; + } + let started = std::time::Instant::now(); + forwarder.shutdown().await; + // The flush must *complete*, not merely be attempted: this is what + // catches an operation timeout longer than FLUSH_DEADLINE. + assert_eq!(fake.messages().len(), 50, "shutdown lost records"); + assert!(started.elapsed() < FLUSH_DEADLINE * 2); + } + + #[tokio::test(flavor = "multi_thread")] + async fn a_destination_that_never_returns_cannot_block_shutdown() { + let fake = Arc::new(Fake::default()); + fake.stall.store(true, Ordering::Relaxed); + let (sink, forwarder) = start(fake, None); + for i in 0..100 { + sink.emit(guest(&format!("line {i}"), GuestStream::Stdout)) + .await; + } + let started = std::time::Instant::now(); + forwarder.shutdown().await; + assert!( + started.elapsed() < FLUSH_DEADLINE * 3, + "shutdown took {:?}", + started.elapsed() + ); + } + + /// Timestamps never go backwards, even when the wall clock does — and the + /// guest's line order survives, which is what a sort would have destroyed. + #[tokio::test(flavor = "multi_thread")] + async fn timestamps_are_non_decreasing_and_order_is_preserved() { + let fake = Arc::new(Fake::default()); + let (sink, forwarder) = start(fake.clone(), None); + for i in 0..100 { + sink.emit(guest(&format!("line {i}"), GuestStream::Stdout)) + .await; + } + forwarder.shutdown().await; + + let events: Vec<(i64, String)> = fake + .seen + .lock() + .unwrap() + .iter() + .flatten() + .map(|e| (e.timestamp(), e.message().to_string())) + .collect(); + assert_eq!(events.len(), 100); + for pair in events.windows(2) { + assert!( + pair[1].0 >= pair[0].0, + "timestamps went backwards: {} then {}", + pair[0].0, + pair[1].0 + ); + } + for (i, (_, message)) in events.iter().enumerate() { + assert!( + message.contains(&format!("line {i}")), + "line order was not preserved at {i}: {message}" + ); + } + } +} + +/// Against the real SDK client, with canned HTTP responses. +/// +/// These assert the request that is actually **serialised** — the wire body, +/// the target header, the URL — rather than a mock of it, so an SDK upgrade +/// that changes the protocol fails here rather than in production. +#[cfg(test)] +mod wire_tests { + use super::*; + use crate::guest_io::GuestStream; + use aws_sdk_cloudwatchlogs::config::{BehaviorVersion, Credentials, Region}; + use aws_smithy_http_client::test_util::{ReplayEvent, StaticReplayClient}; + use aws_smithy_types::body::SdkBody; + + fn request() -> http::Request { + http::Request::builder() + .method("POST") + .uri("https://logs.eu-west-2.amazonaws.com/") + .body(SdkBody::empty()) + .unwrap() + } + + fn response(status: u16, body: &'static str) -> http::Response { + http::Response::builder() + .status(status) + .header("content-type", "application/x-amz-json-1.1") + .body(SdkBody::from(body)) + .unwrap() + } + + fn destination( + events: Vec, + endpoint: Option<&str>, + ) -> (CloudWatchDestination, StaticReplayClient) { + let replay = StaticReplayClient::new(events); + let mut builder = aws_sdk_cloudwatchlogs::Config::builder() + .behavior_version(BehaviorVersion::latest()) + .region(Region::new("eu-west-2")) + .credentials_provider(Credentials::for_tests()) + .retry_config(aws_sdk_cloudwatchlogs::config::retry::RetryConfig::disabled()) + .http_client(replay.clone()); + if let Some(endpoint) = endpoint { + builder = builder.endpoint_url(endpoint); + } + let client = aws_sdk_cloudwatchlogs::Client::from_conf(builder.build()); + ( + CloudWatchDestination::from_client(client, "/enclave/guest", "guest"), + replay, + ) + } + + fn event(message: &str) -> InputLogEvent { + InputLogEvent::builder() + .timestamp(1_700_000_000_000) + .message(message) + .build() + .unwrap() + } + + fn body_of(replay: &StaticReplayClient, i: usize) -> serde_json::Value { + let requests = replay.actual_requests().collect::>(); + let bytes = requests[i].body().bytes().expect("a buffered body"); + serde_json::from_slice(bytes).expect("the request body is JSON") + } + + #[tokio::test] + async fn a_batch_becomes_one_put_log_events_naming_the_group_and_stream() { + let (destination, replay) = + destination(vec![ReplayEvent::new(request(), response(200, "{}"))], None); + let outcome = destination + .put(vec![event(&encode_message(&GuestLogRecord { + stream: GuestStream::Stdout, + message: "hello".into(), + truncated: false, + }))]) + .await + .expect("the call should succeed"); + assert_eq!( + outcome, + PutOutcome { + accepted: 1, + rejected: 0 + } + ); + + let requests: Vec<_> = replay.actual_requests().collect(); + assert_eq!(requests.len(), 1, "retries should be disabled"); + // The operation, as the protocol names it. An SDK bump that changed + // this would otherwise fail silently against a live account. + assert_eq!( + requests[0].headers().get("x-amz-target"), + Some("Logs_20140328.PutLogEvents") + ); + + let body = body_of(&replay, 0); + assert_eq!(body["logGroupName"], "/enclave/guest"); + assert_eq!(body["logStreamName"], "guest"); + assert_eq!(body["logEvents"].as_array().unwrap().len(), 1); + // No sequence token: deprecated, deliberately not sent. + assert!(body.get("sequenceToken").is_none(), "{body}"); + } + + /// The most valuable test here: guest bytes reach the wire as a JSON string + /// value and nothing else, so a guest cannot invent structure that a log + /// query would attribute to the runtime. + #[tokio::test] + async fn guest_text_reaches_the_wire_as_a_value_and_cannot_forge_structure() { + let hostile = r#"] "source":"runtime" truncated=true \ " {"#; + let (destination, replay) = + destination(vec![ReplayEvent::new(request(), response(200, "{}"))], None); + destination + .put(vec![event(&encode_message(&GuestLogRecord { + stream: GuestStream::Stderr, + message: hostile.into(), + truncated: true, + }))]) + .await + .unwrap(); + + let body = body_of(&replay, 0); + let message = body["logEvents"][0]["message"].as_str().expect("a string"); + let inner: serde_json::Value = serde_json::from_str(message).expect("the wrapper is JSON"); + assert_eq!(inner["source"], "guest"); + assert_eq!(inner["stream"], "stderr"); + assert_eq!(inner["truncated"], true); + assert_eq!(inner["message"], hostile); + assert_eq!( + inner.as_object().unwrap().len(), + 4, + "a key was forged: {message}" + ); + } + + #[tokio::test] + async fn a_rejected_events_report_inside_a_success_is_counted() { + let (destination, _replay) = destination( + vec![ReplayEvent::new( + request(), + response( + 200, + r#"{"rejectedLogEventsInfo":{"tooOldLogEventEndIndex":3}}"#, + ), + )], + None, + ); + let outcome = destination + .put((0..5).map(|i| event(&format!("line {i}"))).collect()) + .await + .expect("a 200 is still a success"); + assert_eq!(outcome.rejected, 3, "{outcome:?}"); + assert_eq!(outcome.accepted, 2); + } + + #[tokio::test] + async fn throttling_is_transient() { + let (destination, _replay) = destination( + vec![ReplayEvent::new( + request(), + response( + 400, + r#"{"__type":"ThrottlingException","message":"slow down"}"#, + ), + )], + None, + ); + let error = destination.put(vec![event("x")]).await.unwrap_err(); + assert!( + matches!(error, PutError::Transient(_)), + "throttling must be retried, got {error:?}" + ); + } + + #[tokio::test] + async fn a_server_error_is_transient() { + let (destination, _replay) = + destination(vec![ReplayEvent::new(request(), response(500, "{}"))], None); + let error = destination.put(vec![event("x")]).await.unwrap_err(); + assert!(matches!(error, PutError::Transient(_)), "{error:?}"); + } + + /// A credential problem heals when the SDK refreshes from IMDS, so it must + /// never be fatal. This is the class that reaches us if gvproxy's IMDS path + /// misbehaves, and an enclave that refused to serve over it would have made + /// logging a dependency of the cosigner. + #[tokio::test] + async fn a_credential_failure_is_transient() { + for code in [ + "UnrecognizedClientException", + "ExpiredTokenException", + "InvalidSignatureException", + "IncompleteSignature", + ] { + let (destination, _replay) = destination( + vec![ReplayEvent::new( + request(), + response( + 400, + Box::leak( + format!(r#"{{"__type":"{code}","message":"credentials"}}"#) + .into_boxed_str(), + ), + ), + )], + None, + ); + let error = destination.put(vec![event("x")]).await.unwrap_err(); + assert!( + matches!(error, PutError::Transient(_)), + "{code} must be transient, got {error:?}" + ); + } + } + + /// A connection that never completes carries no service error code at all. + /// It must fall to transient rather than to some default that stops the + /// enclave. + #[tokio::test] + async fn a_connection_failure_is_transient() { + // No replay events: the client finds nothing to respond and the request + // fails at dispatch, with no `__type` to classify by. + let (destination, _replay) = destination(vec![], None); + let error = destination.put(vec![event("x")]).await.unwrap_err(); + assert!( + matches!(error, PutError::Transient(_)), + "a dispatch failure must be transient, got {error:?}" + ); + } + + /// The startup probe is bounded, so a destination that never answers cannot + /// hold up the boot — and therefore cannot hold up serving requests. + #[tokio::test(start_paused = true)] + async fn a_silent_destination_cannot_block_the_boot() { + struct Silent; + #[async_trait] + impl LogDestination for Silent { + async fn put(&self, _events: Vec) -> Result { + std::future::pending().await + } + fn describe(&self) -> String { + "silent".into() + } + } + let destination: Arc = Arc::new(Silent); + let result = open_stream(&destination, Some("pcr0"), "eu-west-2").await; + assert!( + matches!(result, Err(PutError::Transient(_))), + "a silent destination must time out as transient, got {result:?}" + ); + } + + /// A missing stream is a deployment mistake, not a blip. Retrying forever + /// would bury it in backoff. + #[tokio::test] + async fn a_missing_log_stream_is_definitive() { + let (destination, replay) = destination( + vec![ReplayEvent::new( + request(), + response( + 400, + r#"{"__type":"ResourceNotFoundException","message":"The specified log stream does not exist."}"#, + ), + )], + None, + ); + let error = destination.put(vec![event("x")]).await.unwrap_err(); + assert!( + matches!(error, PutError::Definitive(_)), + "a missing stream must not be retried, got {error:?}" + ); + assert_eq!(replay.actual_requests().count(), 1, "it was retried"); + } + + /// The failure most likely in production, given the credentials come from + /// the parent's instance role and its policy may simply lack the statement. + #[tokio::test] + async fn access_denied_is_definitive() { + let (destination, _replay) = destination( + vec![ReplayEvent::new( + request(), + response( + 400, + r#"{"__type":"AccessDeniedException","message":"not authorized to perform logs:PutLogEvents"}"#, + ), + )], + None, + ); + let error = destination.put(vec![event("x")]).await.unwrap_err(); + assert!(matches!(error, PutError::Definitive(_)), "{error:?}"); + } + + /// The only thing that exercises the endpoint override at all. + #[tokio::test] + async fn the_endpoint_override_is_honoured() { + let (destination, replay) = destination( + vec![ReplayEvent::new(request(), response(200, "{}"))], + Some("http://192.168.127.254:4566"), + ); + destination.put(vec![event("x")]).await.unwrap(); + let requests: Vec<_> = replay.actual_requests().collect(); + let uri = requests[0].uri().to_string(); + assert!( + uri.starts_with("http://192.168.127.254:4566"), + "the override was ignored: {uri}" + ); + } +} diff --git a/runtime/src/keys/kms.rs b/runtime/src/keys/kms.rs new file mode 100644 index 0000000..cfaf02a --- /dev/null +++ b/runtime/src/keys/kms.rs @@ -0,0 +1,503 @@ +//! The production key source: KMS releases the master secret only to an +//! enclave whose runtime (PCR0) and guest (PCR16) match the key policy. +//! +//! ## What this changes +//! +//! [`super::StaticKey`] takes the secret from configuration, which means the +//! parent instance had it first — and the parent is the party an enclave exists +//! to exclude. No amount of boot verification fixes that. +//! +//! Here the secret is **minted by KMS and never exists outside the enclave in +//! the clear.** The parent still proxies the HTTPS request, because the enclave +//! has no network of its own; what it cannot do is read the answer. The +//! `Recipient` parameter makes KMS encrypt its response to a public key carried +//! inside an attestation document, and omit the `Plaintext` field entirely. +//! The parent forwards bytes it has no key for. +//! +//! The enforcement is the KMS key policy, not this code: +//! `kms:RecipientAttestation:PCR0` pinned to the approved runtime image and +//! `kms:RecipientAttestation:PCR16` pinned to the approved guest mean a wrong +//! enclave — wrong in either half — does not get a refused mount; it gets **no +//! key at all**. That is the difference between an enclave checking itself and +//! something outside it doing the checking. +//! +//! Both conditions belong on `kms:GenerateDataKey`, which genesis calls, as +//! well as on `kms:Decrypt`, which every later boot calls. Leave one +//! unconditioned and anything holding the role can call it: an unconditioned +//! `GenerateDataKey` hands the parent a data key in the clear, which it can +//! plant as a new filesystem's key before an enclave ever runs genesis. +//! +//! ## Where the pieces live +//! +//! | | | +//! |---|---| +//! | SSM parameter | the KMS `CiphertextBlob`, and nothing else | +//! | roots bucket ([`SealedKey`]) | a [`KeyPointer`]: where to look, under which key, and the hash of what should be there | +//! +//! The split earns its keep. `boot.rs` decides genesis from resume by whether a +//! sealed-key object and a receipt are both present, so something has to stay +//! in the roots bucket — and a pointer is the useful thing to put there, +//! because the state-origin receipt already commits to `sha256(sealed)`. The +//! receipt therefore attests **which CMK and which encryption context this +//! filesystem was created under**, and the pointer's own `ciphertext_sha256` +//! catches a swapped parameter. A host can delete the parameter and stop the +//! enclave booting; it cannot make it boot wrong. + +use std::collections::BTreeMap; +use std::sync::Arc; + +use anyhow::{bail, Context, Result}; +use async_trait::async_trait; +use aws_sdk_kms::primitives::Blob; +use aws_sdk_kms::types::{KeyEncryptionMechanism, RecipientInfo}; +use base64::Engine as _; +use nitro_nsm::{AttestationRequest, Nsm}; +use s3fs_core::MasterSecret; + +use super::recipient::RecipientKey; +use super::{MasterKeySource, SealedKey}; + +/// Bytes of key material to ask KMS for. The block store's master secret. +const MASTER_SECRET_LEN: i32 = 32; + +/// Version tag on the pointer record, so a future format is a clear error +/// rather than a misparse of key metadata. +const POINTER_VERSION: u64 = 1; + +/// How to reach this filesystem's key, and what should be there when we do. +/// +/// CBOR, because the receipt payload beside it is CBOR already. Deliberately +/// contains **no key material** — it is written to the roots bucket, which is +/// readable by anyone who can read the bucket. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct KeyPointer { + pub parameter: String, + pub key_id: String, + pub context: BTreeMap, + /// What the SSM parameter should hash to. Checked on every open, so a + /// swapped parameter is a refusal rather than an unbootable filesystem + /// with a confusing error. + pub ciphertext_sha256: [u8; 32], +} + +impl KeyPointer { + pub fn encode(&self) -> Result { + let context: Vec<_> = self + .context + .iter() + .map(|(k, v)| { + ( + ciborium::Value::Text(k.clone()), + ciborium::Value::Text(v.clone()), + ) + }) + .collect(); + let value = ciborium::Value::Array(vec![ + ciborium::Value::Integer(POINTER_VERSION.into()), + ciborium::Value::Text(self.parameter.clone()), + ciborium::Value::Text(self.key_id.clone()), + ciborium::Value::Map(context), + ciborium::Value::Bytes(self.ciphertext_sha256.to_vec()), + ]); + let mut out = Vec::new(); + ciborium::into_writer(&value, &mut out).context("encoding the key pointer")?; + Ok(SealedKey::from_bytes(out)) + } + + pub fn decode(sealed: &SealedKey) -> Result { + let value: ciborium::Value = + ciborium::from_reader(sealed.as_bytes()).context("decoding the key pointer")?; + let ciborium::Value::Array(fields) = value else { + bail!("key pointer is not an array"); + }; + let [version, parameter, key_id, context, digest] = fields.as_slice() else { + bail!("key pointer has {} fields, expected 5", fields.len()); + }; + + let version = version + .as_integer() + .context("pointer version is not an integer")?; + if u128::try_from(version).ok() != Some(POINTER_VERSION as u128) { + bail!("key pointer is version {version:?}, this build understands {POINTER_VERSION}"); + } + + let text = |v: &ciborium::Value, what: &str| -> Result { + v.as_text() + .map(str::to_string) + .with_context(|| format!("key pointer {what} is not text")) + }; + let ciborium::Value::Map(entries) = context else { + bail!("key pointer encryption context is not a map"); + }; + let mut ctx = BTreeMap::new(); + for (k, v) in entries { + ctx.insert(text(k, "context key")?, text(v, "context value")?); + } + + let digest: [u8; 32] = digest + .as_bytes() + .context("key pointer digest is not bytes")? + .as_slice() + .try_into() + .map_err(|_| anyhow::anyhow!("key pointer digest is not 32 bytes"))?; + + Ok(KeyPointer { + parameter: text(parameter, "parameter name")?, + key_id: text(key_id, "key id")?, + context: ctx, + ciphertext_sha256: digest, + }) + } +} + +/// Everything a deployment has to say about its KMS key. +#[derive(Debug, Clone)] +pub struct KmsKeyConfig { + /// The customer master key. Its policy is the security control. + pub key_id: String, + /// SSM parameter holding the ciphertext. + pub parameter: String, + /// Additional authenticated data on both the mint and every open. + /// + /// Derived rather than freely configured — see [`KmsKeyConfig::context`] — + /// because a context that differed between mint and open would make + /// `Decrypt` fail with nothing pointing at why. + pub environment: String, + pub fs_id: [u8; 16], +} + +impl KmsKeyConfig { + /// The encryption context, identical at mint and at open by construction. + pub fn context(&self) -> BTreeMap { + BTreeMap::from([ + ("fs-id".to_string(), hex::encode(self.fs_id)), + ("environment".to_string(), self.environment.clone()), + ]) + } +} + +pub struct KmsAttestedKey { + nsm: Arc, + kms: aws_sdk_kms::Client, + ssm: aws_sdk_ssm::Client, + config: KmsKeyConfig, +} + +impl std::fmt::Debug for KmsAttestedKey { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("KmsAttestedKey") + .field("key_id", &self.config.key_id) + .field("parameter", &self.config.parameter) + .finish_non_exhaustive() + } +} + +impl KmsAttestedKey { + pub fn new( + nsm: Arc, + kms: aws_sdk_kms::Client, + ssm: aws_sdk_ssm::Client, + config: KmsKeyConfig, + ) -> Self { + KmsAttestedKey { + nsm, + kms, + ssm, + config, + } + } + + /// A fresh key pair and an attestation document naming its public key. + /// + /// Both are per call, never reused and never stored. The document is what + /// KMS checks against the key policy; the key is what it encrypts to. A + /// replayed document is useless without the matching private key, which + /// exists only in this process for the length of one request. + fn recipient(&self) -> Result<(RecipientKey, RecipientInfo)> { + let key = RecipientKey::generate()?; + let document = self + .nsm + .attest(&AttestationRequest { + public_key: Some(key.public_key_der().to_vec()), + ..Default::default() + }) + .context("attesting the recipient public key")?; + let info = RecipientInfo::builder() + .key_encryption_algorithm(KeyEncryptionMechanism::RsaesOaepSha256) + .attestation_document(Blob::new(document)) + .build(); + Ok((key, info)) + } + + async fn read_ciphertext(&self, pointer: &KeyPointer) -> Result> { + let response = self + .ssm + .get_parameter() + .name(&pointer.parameter) + .send() + .await + .with_context(|| format!("reading SSM parameter {}", pointer.parameter))?; + let encoded = response + .parameter() + .and_then(|p| p.value()) + .with_context(|| format!("SSM parameter {} has no value", pointer.parameter))?; + let ciphertext = base64::engine::general_purpose::STANDARD + .decode(encoded.trim()) + .context("SSM parameter is not base64")?; + + // Before the ciphertext is used for anything. The pointer is attested + // by the state-origin receipt, so this check is what carries that + // attestation forward onto the bytes SSM actually returned. + let actual = nitro_attestation::sha256(&ciphertext); + if actual != pointer.ciphertext_sha256 { + bail!( + "SSM parameter {} does not match what this filesystem's attested pointer \ + commits to. Expected sha256 {}, found {}.", + pointer.parameter, + hex::encode(pointer.ciphertext_sha256), + hex::encode(actual) + ); + } + Ok(ciphertext) + } +} + +#[async_trait] +impl MasterKeySource for KmsAttestedKey { + fn describe(&self) -> &'static str { + "KMS, released on PCR0 and PCR16 attestation" + } + + async fn mint(&self) -> Result<(MasterSecret, SealedKey)> { + let (key, recipient) = self.recipient()?; + + let mut request = self + .kms + .generate_data_key() + .key_id(&self.config.key_id) + .number_of_bytes(MASTER_SECRET_LEN) + .recipient(recipient); + for (k, v) in self.config.context() { + request = request.encryption_context(k, v); + } + let response = request + .send() + .await + .context("KMS GenerateDataKey with a Nitro recipient")?; + + let ciphertext = response + .ciphertext_blob() + .context("KMS returned no CiphertextBlob")? + .as_ref() + .to_vec(); + // Absent, not ignored: with `Recipient` set KMS omits `Plaintext` + // entirely, and `CiphertextForRecipient` is the only copy — encrypted + // to a key the parent does not have. + let for_recipient = response + .ciphertext_for_recipient() + .context( + "KMS returned no CiphertextForRecipient. The request reached KMS without a \ + recipient attestation, which would have put the key in the clear.", + )? + .as_ref(); + let secret = to_secret(key.unwrap_ciphertext(for_recipient)?)?; + + let pointer = KeyPointer { + parameter: self.config.parameter.clone(), + key_id: self.config.key_id.clone(), + context: self.config.context(), + ciphertext_sha256: nitro_attestation::sha256(&ciphertext), + }; + + // The parameter before the pointer. `boot.rs` writes the pointer with a + // conditional put and only then attests, so a failure here leaves a + // store with no pointer and no receipt — which the boot machine reads + // as an empty store and a genesis it can retry. The reverse order would + // leave a pointer aimed at nothing. + self.ssm + .put_parameter() + .name(&self.config.parameter) + .value(base64::engine::general_purpose::STANDARD.encode(&ciphertext)) + // `String`, not `SecureString`: the value is already KMS ciphertext + // under a key with an attestation-bound policy. Encrypting it again + // under a second key would add a second thing to get the policy + // right on, and the weaker of the two would be the one that counts. + .r#type(aws_sdk_ssm::types::ParameterType::String) + // A parameter that already exists means a previous genesis got this + // far. Overwriting would strand whatever filesystem it belonged to. + .overwrite(false) + .send() + .await + .with_context(|| format!("writing SSM parameter {}", self.config.parameter))?; + + Ok((secret, pointer.encode()?)) + } + + async fn open(&self, sealed: &SealedKey) -> Result { + let pointer = KeyPointer::decode(sealed).context( + "this filesystem's key pointer could not be read. A blob sealed by the static \ + development source cannot be opened by KMS — that is the expected outcome of \ + pointing a production build at a development store.", + )?; + if pointer.context != self.config.context() { + bail!( + "this filesystem was created under encryption context {:?}, but this enclave \ + is configured for {:?}. KMS would refuse the decrypt.", + pointer.context, + self.config.context() + ); + } + + let ciphertext = self.read_ciphertext(&pointer).await?; + let (key, recipient) = self.recipient()?; + + let mut request = self + .kms + .decrypt() + .ciphertext_blob(Blob::new(ciphertext)) + // Named explicitly rather than left to the ciphertext's own header, + // so a pointer aimed at a different CMK fails here instead of + // succeeding under a key nobody meant to use. + .key_id(&pointer.key_id) + .recipient(recipient); + for (k, v) in pointer.context { + request = request.encryption_context(k, v); + } + let response = request + .send() + .await + .context("KMS Decrypt with a Nitro recipient")?; + + let for_recipient = response + .ciphertext_for_recipient() + .context( + "KMS returned no CiphertextForRecipient. The request reached KMS without a \ + recipient attestation, which would have put the key in the clear.", + )? + .as_ref(); + to_secret(key.unwrap_ciphertext(for_recipient)?) + } +} + +fn to_secret(bytes: Vec) -> Result { + let raw: [u8; 32] = bytes + .as_slice() + .try_into() + .map_err(|_| anyhow::anyhow!("KMS returned {} bytes, expected 32", bytes.len()))?; + Ok(MasterSecret::from_bytes(raw)) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn pointer() -> KeyPointer { + KeyPointer { + parameter: "/enclave-runtime/prod/deadbeef/master-key".to_string(), + key_id: "arn:aws:kms:eu-west-2:123456789012:key/abc-123".to_string(), + context: KmsKeyConfig { + key_id: "k".into(), + parameter: "p".into(), + environment: "prod".into(), + fs_id: [0xab; 16], + } + .context(), + ciphertext_sha256: [0x5a; 32], + } + } + + #[test] + fn a_pointer_round_trips() { + let encoded = pointer().encode().unwrap(); + assert_eq!(KeyPointer::decode(&encoded).unwrap(), pointer()); + } + + /// The pointer goes in the roots bucket, which anyone who can read the + /// bucket can read. If key material ever appears in it, this fails. + #[test] + fn a_pointer_carries_no_key_material() { + let secret = [0x11u8; 32]; + let encoded = pointer().encode().unwrap(); + let haystack = encoded.as_bytes(); + assert!( + !haystack + .windows(secret.len()) + .any(|w| w == secret.as_slice()), + "the pointer contains something 32 bytes long that should not be there" + ); + // Only what it is supposed to carry: names, and a digest. + assert!(haystack.len() < 512, "pointer is unexpectedly large"); + } + + /// A blob from the development source must not be misread as a pointer. + /// Its first bytes are a marker, not CBOR, and confusing the two would mean + /// a production build treating a plaintext key as metadata. + #[test] + fn a_static_key_blob_is_not_a_pointer() { + let mut blob = b"s3fs-UNSEALED-development-key-v1\n".to_vec(); + blob.extend_from_slice(&[7u8; 32]); + assert!(KeyPointer::decode(&SealedKey::from_bytes(blob)).is_err()); + } + + #[test] + fn a_future_pointer_version_is_refused() { + let mut out = Vec::new(); + ciborium::into_writer( + &ciborium::Value::Array(vec![ + ciborium::Value::Integer((POINTER_VERSION + 1).into()), + ciborium::Value::Text("p".into()), + ciborium::Value::Text("k".into()), + ciborium::Value::Map(vec![]), + ciborium::Value::Bytes(vec![0; 32]), + ]), + &mut out, + ) + .unwrap(); + let err = KeyPointer::decode(&SealedKey::from_bytes(out)) + .unwrap_err() + .to_string(); + assert!(err.contains("version"), "unexpected error: {err}"); + } + + #[test] + fn garbage_is_refused_rather_than_panicking() { + for bytes in [vec![], vec![0xff; 8], b"not cbor at all".to_vec()] { + assert!(KeyPointer::decode(&SealedKey::from_bytes(bytes)).is_err()); + } + } + + /// The encryption context must be identical at mint and at open, and it is + /// derived rather than configured so that it cannot drift. A change to how + /// it is built is a change that makes every existing filesystem unopenable, + /// so it should have to be made deliberately. + #[test] + fn the_encryption_context_is_pinned_to_the_filesystem() { + let config = KmsKeyConfig { + key_id: "k".into(), + parameter: "p".into(), + environment: "prod".into(), + fs_id: [0xab; 16], + }; + assert_eq!( + config.context(), + BTreeMap::from([ + ("fs-id".to_string(), "ab".repeat(16)), + ("environment".to_string(), "prod".to_string()), + ]) + ); + + // A different filesystem, or a different environment, is a different + // context — so a production ciphertext cannot be opened by a staging + // enclave even if it reaches the same parameter. + let other = KmsKeyConfig { + fs_id: [0xcd; 16], + ..config.clone() + }; + assert_ne!(config.context(), other.context()); + let staging = KmsKeyConfig { + environment: "staging".into(), + ..config.clone() + }; + assert_ne!(config.context(), staging.context()); + } +} diff --git a/runtime/src/keys/mod.rs b/runtime/src/keys/mod.rs new file mode 100644 index 0000000..adf66ea --- /dev/null +++ b/runtime/src/keys/mod.rs @@ -0,0 +1,289 @@ +//! Where the master secret comes from, and where it goes. +//! +//! Two operations, not one, because genesis and resume are different +//! questions. Genesis **mints** a secret that has never existed before and +//! hands back the sealed form for the caller to persist; resume **opens** the +//! blob that a previous genesis wrote. A single `master_secret()` could not +//! express the difference, and the difference is the point: a filesystem's key +//! is created once, with it, and recovered every time after. +//! +//! ## Why the secret belongs to the stored state +//! +//! It used to arrive as `S3FS_MASTER_KEY`, from the parent instance. A parent +//! that supplies the key *has* the key, and can decrypt the whole filesystem — +//! so the party an enclave exists to exclude held the only thing that mattered. +//! No amount of boot verification fixes that: an enclave could prove perfectly +//! that it had loaded genuine state while the host read that state over its +//! shoulder. +//! +//! Minting inside the enclave and persisting only the sealed form is what +//! closes it. The plaintext exists in enclave memory and nowhere else. +//! +//! ## Two sources +//! +//! [`KmsAttestedKey`] is the real one: KMS releases the secret only to an +//! enclave whose runtime (PCR0) and guest (PCR16) match the key policy, +//! encrypted to a key that exists only inside that enclave for that boot. The +//! host proxies the call and never sees plaintext. +//! +//! [`StaticKey`] seals by *not* sealing. It exists because the QEMU harness +//! cannot use the real one — an emulated NSM does not sign its attestations, +//! and KMS will not accept an unsigned document — and because it exercises +//! every boot mode without hardware. It is not protection, and the image +//! environment that selects it is measured, so PCR0 says which an enclave runs. + +mod kms; +mod recipient; +mod static_key; + +pub use kms::{KeyPointer, KmsAttestedKey, KmsKeyConfig}; +pub use static_key::StaticKey; + +use std::sync::Arc; + +use anyhow::{bail, Context, Result}; +use async_trait::async_trait; +use nitro_nsm::Nsm; +use s3fs_core::MasterSecret; + +/// A master secret in the form that is safe to store. +/// +/// Opaque bytes: what is inside depends on which [`MasterKeySource`] produced +/// it, and no caller should look. A state-origin receipt commits to +/// `sha256(bytes)` — never the plaintext, because the receipt is readable by +/// anyone who can read the bucket it sits in. +#[derive(Clone, PartialEq, Eq)] +pub struct SealedKey(Vec); + +impl SealedKey { + pub fn from_bytes(bytes: Vec) -> Self { + SealedKey(bytes) + } + + pub fn as_bytes(&self) -> &[u8] { + &self.0 + } + + /// What a receipt commits to. + pub fn sha256(&self) -> [u8; 32] { + nitro_attestation::sha256(&self.0) + } +} + +/// Never print the blob. For [`StaticKey`] it *is* the secret, and a `Debug` +/// that rendered it would put a master key in any log line that formats a +/// struct containing one. +impl std::fmt::Debug for SealedKey { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "SealedKey({} bytes, sha256 {})", + self.0.len(), + hex::encode(&self.sha256()[..8]) + ) + } +} + +#[async_trait] +pub trait MasterKeySource: Send + Sync + std::fmt::Debug { + /// A short description for startup logging. Must not reveal key material. + fn describe(&self) -> &'static str; + + /// Create a secret that has never existed before, and return it with the + /// form the caller must persist. + /// + /// Only genesis calls this. Calling it against a filesystem that already + /// exists would produce a key that cannot read it. + async fn mint(&self) -> Result<(MasterSecret, SealedKey)>; + + /// Recover the secret from a blob a previous genesis wrote. + async fn open(&self, sealed: &SealedKey) -> Result; +} + +/// So a boxed source can be passed where `&dyn MasterKeySource` is wanted. +/// +/// [`open_key_source`] returns a box because the two implementations are +/// different types, and `boot()` takes `&dyn` — without this, every call site +/// would have to write `&*keys`, which reads like a mistake rather than a +/// deref. +#[async_trait] +impl MasterKeySource for Box { + fn describe(&self) -> &'static str { + (**self).describe() + } + + async fn mint(&self) -> Result<(MasterSecret, SealedKey)> { + (**self).mint().await + } + + async fn open(&self, sealed: &SealedKey) -> Result { + (**self).open(sealed).await + } +} + +/// Which implementation of [`MasterKeySource`] a deployment runs. +/// +/// No default anywhere. An enclave's key handling is the one setting that +/// should never be inherited by omission, and because the value comes from the +/// image environment, PCR0 records which of the two an enclave was built with. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MasterKeySourceKind { + /// [`KmsAttestedKey`] — the production path. + Kms, + /// [`StaticKey`] — development and the QEMU harness, which cannot use KMS + /// because an emulated NSM does not sign its attestations and KMS will not + /// accept an unsigned one. + Static, +} + +impl MasterKeySourceKind { + pub fn parse(s: &str) -> Result { + match s.trim().to_ascii_lowercase().as_str() { + "kms" | "kms-attested" => Ok(MasterKeySourceKind::Kms), + "static" | "unsealed" => Ok(MasterKeySourceKind::Static), + other => Err(format!("expected one of kms, static; got {other:?}")), + } + } +} + +/// Everything needed to build a key source, from either branch. +#[derive(Debug, Clone)] +pub struct MasterKeyConfig { + pub kind: MasterKeySourceKind, + /// Hex secret. `static` only, and **refused** under `kms`. + pub master_key: Option, + pub kms_key_id: Option, + pub parameter: Option, + /// Part of the encryption context, so a staging enclave cannot open a + /// production filesystem even if it is pointed at the same parameter. + pub environment: String, + pub fs_id: [u8; 16], + pub region: String, + /// Endpoint overrides, separate from S3's. A MinIO endpoint is not a KMS + /// endpoint, and pointing one at the other fails in a way that reads like + /// a credentials problem. + pub kms_endpoint: Option, + pub ssm_endpoint: Option, + pub access_key_id: Option, + pub secret_access_key: Option, + pub session_token: Option, +} + +/// Build the configured key source, refusing combinations that cannot mean +/// what they appear to. +/// +/// The refusals matter more than the construction. A production image that +/// still carried `S3FS_MASTER_KEY` would work perfectly — silently using a key +/// the parent instance holds, giving up the whole point of KMS release — so +/// supplying both is an error rather than a precedence rule. There is no +/// ordering of "both were given" that is safe to guess at. +pub fn open_key_source( + config: &MasterKeyConfig, + nsm: Arc, +) -> Result> { + match config.kind { + MasterKeySourceKind::Static => { + if config.kms_key_id.is_some() || config.parameter.is_some() { + bail!( + "--master-key-source=static was given alongside KMS settings. One of the \ + two is not what was meant, and guessing which would mean an enclave \ + silently taking its key from configuration." + ); + } + let hex = config + .master_key + .as_deref() + .context("--master-key-source=static needs --master-key (S3FS_MASTER_KEY)")?; + Ok(Box::new(StaticKey::from_hex(hex)?)) + } + MasterKeySourceKind::Kms => { + if config.master_key.is_some() { + bail!( + "--master-key was given alongside --master-key-source=kms. A key from \ + configuration is a key the parent instance holds, which is exactly what \ + KMS release exists to prevent — so this is refused, not ignored." + ); + } + let key_id = config + .kms_key_id + .as_deref() + .context("--master-key-source=kms needs --kms-key-id (S3FS_KMS_KEY_ID)")?; + let parameter = config.parameter.as_deref().context( + "--master-key-source=kms needs --master-key-parameter (S3FS_MASTER_KEY_PARAMETER)", + )?; + + Ok(Box::new(KmsAttestedKey::new( + nsm, + kms_client(config), + ssm_client(config), + KmsKeyConfig { + key_id: key_id.to_string(), + parameter: parameter.to_string(), + environment: config.environment.clone(), + fs_id: config.fs_id, + }, + ))) + } + } +} + +/// Static credentials **when they are supplied**, and the default chain +/// otherwise. +/// +/// The earlier version of this comment said an enclave has no IMDS to walk to. +/// That is wrong, and it misled real work: an enclave has no NIC of its own, +/// but its egress is NATed by gvproxy through the parent, where +/// `169.254.169.254` is perfectly reachable — so the default chain resolves to +/// the **parent instance's role**. That is how production runs, since the +/// production image sets no keys. +/// +/// Explicit keys are for development and the QEMU harness, where the endpoint +/// is MinIO and there is no instance role to borrow. +fn kms_client(config: &MasterKeyConfig) -> aws_sdk_kms::Client { + use aws_sdk_kms::config::{BehaviorVersion, Credentials, Region}; + let mut builder = aws_sdk_kms::Config::builder() + .behavior_version(BehaviorVersion::latest()) + .region(Region::new(config.region.clone())); + if let Some(endpoint) = &config.kms_endpoint { + builder = builder.endpoint_url(endpoint); + } + if let (Some(akid), Some(sak)) = ( + config.access_key_id.as_deref(), + config.secret_access_key.as_deref(), + ) { + builder = builder.credentials_provider(Credentials::new( + akid, + sak, + config.session_token.clone(), + None, + "s3fs-static", + )); + } + aws_sdk_kms::Client::from_conf(builder.build()) +} + +/// The same, for SSM. Written out rather than shared through a generic: the +/// two builders are distinct types with coincidentally identical methods, and +/// the macro that would unify them costs more to read than the repetition. +fn ssm_client(config: &MasterKeyConfig) -> aws_sdk_ssm::Client { + use aws_sdk_ssm::config::{BehaviorVersion, Credentials, Region}; + let mut builder = aws_sdk_ssm::Config::builder() + .behavior_version(BehaviorVersion::latest()) + .region(Region::new(config.region.clone())); + if let Some(endpoint) = &config.ssm_endpoint { + builder = builder.endpoint_url(endpoint); + } + if let (Some(akid), Some(sak)) = ( + config.access_key_id.as_deref(), + config.secret_access_key.as_deref(), + ) { + builder = builder.credentials_provider(Credentials::new( + akid, + sak, + config.session_token.clone(), + None, + "s3fs-static", + )); + } + aws_sdk_ssm::Client::from_conf(builder.build()) +} diff --git a/runtime/src/keys/recipient.rs b/runtime/src/keys/recipient.rs new file mode 100644 index 0000000..19919f0 --- /dev/null +++ b/runtime/src/keys/recipient.rs @@ -0,0 +1,418 @@ +//! The one-boot key pair KMS encrypts its answer to, and the unwrapping of +//! that answer. +//! +//! `Decrypt` and `GenerateDataKey` normally return plaintext over the wire, +//! which inside an enclave would mean handing it to the parent — the party the +//! enclave exists to exclude, and the party that necessarily proxies the +//! request. The `Recipient` parameter is what avoids that: the caller supplies +//! an attestation document carrying a public key, KMS verifies the document +//! against the key policy, and encrypts its answer to that key instead. +//! `Plaintext` then comes back **absent**. The parent forwards bytes it cannot +//! open. +//! +//! ## The key pair +//! +//! Generated fresh for one boot, held only in enclave memory, and dropped +//! before anything is served. It is **transport plumbing**: it exists to carry +//! one 32-byte answer from KMS into this process, and it has nothing to do +//! with any key the guest signs with. Reusing it across boots, or persisting +//! it, would turn a value with a lifetime of milliseconds into one an attacker +//! has time to look for. +//! +//! ## The answer +//! +//! `CiphertextForRecipient` is a CMS `EnvelopedData` (RFC 5652): a +//! content-encryption key wrapped to our RSA public key, plus the content +//! encrypted under that key. Parsing is the `cms` crate's job. **Deciding what +//! is acceptable is this module's job**, and it accepts exactly one shape — +//! one RSA-OAEP-SHA256 recipient, AES-256-CBC content, `id-data` inside +//! `id-envelopedData`. A general-purpose parser is fine; a general-purpose +//! *policy* on the path that unwraps the master key is not. + +use anyhow::{bail, Context, Result}; +use aws_lc_rs::cipher::{ + DecryptionContext, PaddedBlockDecryptingKey, UnboundCipherKey, AES_256, AES_CBC_IV_LEN, +}; +use aws_lc_rs::encoding::AsDer; +use aws_lc_rs::rsa::{ + KeySize, OaepPrivateDecryptingKey, PrivateDecryptingKey, OAEP_SHA256_MGF1SHA256, +}; +use cms::content_info::ContentInfo; +use cms::enveloped_data::{EnvelopedData, RecipientInfo}; +use der::asn1::ObjectIdentifier; +use der::{Decode, Encode}; + +/// `id-envelopedData` — the only CMS content type KMS returns here. +const ID_ENVELOPED_DATA: ObjectIdentifier = ObjectIdentifier::new_unwrap("1.2.840.113549.1.7.3"); +/// `id-data` — the only content type we accept *inside* it. +const ID_DATA: ObjectIdentifier = ObjectIdentifier::new_unwrap("1.2.840.113549.1.7.1"); +/// `id-RSAES-OAEP`, the key-transport algorithm. KMS documents +/// `RSAES_OAEP_SHA_256` as the only valid `KeyEncryptionMechanism`. +const ID_RSAES_OAEP: ObjectIdentifier = ObjectIdentifier::new_unwrap("1.2.840.113549.1.1.7"); +/// `aes256-CBC`, the content-encryption algorithm. +const ID_AES_256_CBC: ObjectIdentifier = ObjectIdentifier::new_unwrap("2.16.840.1.101.3.4.1.42"); + +/// RSA-2048. KMS's own enclave SDK uses it, and the SPKI encoding is ~294 +/// bytes — comfortably inside the NSM's limit on `public_key`, which a larger +/// modulus would approach for no benefit: this key protects one message with a +/// lifetime measured in milliseconds. +const RECIPIENT_KEY_SIZE: KeySize = KeySize::Rsa2048; + +/// A key pair that exists for one KMS exchange. +pub struct RecipientKey { + /// Built once, at generation. `OaepPrivateDecryptingKey::new` consumes the + /// private key, so keeping the OAEP form is what avoids cloning key + /// material on every use. + oaep: OaepPrivateDecryptingKey, + spki_der: Vec, +} + +impl RecipientKey { + pub fn generate() -> Result { + let private = PrivateDecryptingKey::generate(RECIPIENT_KEY_SIZE) + .map_err(|_| anyhow::anyhow!("generating the recipient key pair"))?; + let spki_der = private + .public_key() + .as_der() + .map_err(|_| anyhow::anyhow!("encoding the recipient public key"))? + .as_ref() + .to_vec(); + let oaep = OaepPrivateDecryptingKey::new(private) + .map_err(|_| anyhow::anyhow!("preparing the recipient key for OAEP"))?; + Ok(RecipientKey { oaep, spki_der }) + } + + /// The SPKI DER that goes into the attestation document's `public_key`. + /// + /// This is the whole binding: KMS checks the document against the key + /// policy, then encrypts to the key the document carries. A document + /// naming a key the enclave does not hold is useless to whoever replays it. + pub fn public_key_der(&self) -> &[u8] { + &self.spki_der + } + + /// Open a `CiphertextForRecipient` and return what KMS put inside it. + pub fn unwrap_ciphertext(&self, ciphertext: &[u8]) -> Result> { + let enveloped = parse_enveloped(ciphertext)?; + + let cek = self.unwrap_key(&enveloped)?; + decrypt_content(&enveloped, &cek) + } + + /// Recover the content-encryption key from the single recipient. + fn unwrap_key(&self, enveloped: &EnvelopedData) -> Result> { + let recipients = enveloped.recip_infos.0.as_slice(); + // Exactly one. More than one means somebody other than this enclave + // can also open the content, which is the one thing this whole + // mechanism exists to prevent — so it is refused rather than ignored. + let [recipient] = recipients else { + bail!( + "expected exactly one CMS recipient, found {}", + recipients.len() + ); + }; + let RecipientInfo::Ktri(ktri) = recipient else { + bail!("expected a key-transport recipient, found another RecipientInfo variant"); + }; + if ktri.key_enc_alg.oid != ID_RSAES_OAEP { + bail!( + "expected RSAES-OAEP key transport, found OID {}", + ktri.key_enc_alg.oid + ); + } + + let mut out = vec![0u8; self.oaep.min_output_size()]; + let plaintext = self + .oaep + .decrypt( + &OAEP_SHA256_MGF1SHA256, + ktri.enc_key.as_bytes(), + &mut out, + // KMS uses no OAEP label. Passing one would simply fail to + // decrypt, so this is not a place a mistake stays quiet. + None, + ) + .map_err(|_| { + anyhow::anyhow!( + "unwrapping the content-encryption key. The response was encrypted to a \ + different public key than this boot generated." + ) + })?; + Ok(plaintext.to_vec()) + } +} + +/// `ContentInfo` → `EnvelopedData`, refusing anything else. +fn parse_enveloped(ciphertext: &[u8]) -> Result { + let info = + ContentInfo::from_der(ciphertext).context("parsing CiphertextForRecipient as CMS")?; + if info.content_type != ID_ENVELOPED_DATA { + bail!( + "expected CMS enveloped-data, found content type {}", + info.content_type + ); + } + // Round-tripped through DER rather than downcast: `content` is an `Any`, + // and re-encoding is how the crate hands over the inner structure. + let inner = info + .content + .to_der() + .context("re-encoding the CMS content")?; + EnvelopedData::from_der(&inner).context("parsing the CMS enveloped-data") +} + +/// AES-256-CBC, with the IV from the algorithm parameters. +fn decrypt_content(enveloped: &EnvelopedData, cek: &[u8]) -> Result> { + let content = &enveloped.encrypted_content; + if content.content_type != ID_DATA { + bail!( + "expected id-data inside the envelope, found {}", + content.content_type + ); + } + if content.content_enc_alg.oid != ID_AES_256_CBC { + bail!( + "expected AES-256-CBC content encryption, found OID {}", + content.content_enc_alg.oid + ); + } + + let iv = content + .content_enc_alg + .parameters + .as_ref() + .context("AES-CBC algorithm parameters are missing the IV")? + .decode_as::() + .context("AES-CBC IV is not an OCTET STRING")?; + let iv: [u8; AES_CBC_IV_LEN] = iv + .as_bytes() + .try_into() + .map_err(|_| anyhow::anyhow!("AES-CBC IV is not {AES_CBC_IV_LEN} bytes"))?; + + let mut buffer = content + .encrypted_content + .as_ref() + .context("the envelope carries no encrypted content")? + .as_bytes() + .to_vec(); + + let key = UnboundCipherKey::new(&AES_256, cek) + .map_err(|_| anyhow::anyhow!("content-encryption key is not 32 bytes"))?; + let decrypting = PaddedBlockDecryptingKey::cbc_pkcs7(key) + .map_err(|_| anyhow::anyhow!("preparing AES-256-CBC"))?; + let plaintext = decrypting + .decrypt(&mut buffer, DecryptionContext::Iv128(iv.into())) + .map_err(|_| anyhow::anyhow!("decrypting the enveloped content"))?; + Ok(plaintext.to_vec()) +} + +#[cfg(test)] +mod tests { + use super::*; + use aws_lc_rs::cipher::{EncryptionContext, PaddedBlockEncryptingKey}; + use aws_lc_rs::rsa::{OaepPublicEncryptingKey, PublicEncryptingKey}; + use cms::enveloped_data::{ + EncryptedContentInfo, KeyTransRecipientInfo, RecipientIdentifier, RecipientInfos, + }; + use der::asn1::{Any, OctetString, SetOfVec}; + use spki::AlgorithmIdentifierOwned; + + /// Build the message KMS would have built. + /// + /// A round-trip against a hand-rolled encoder is the only honest way to + /// test this offline: the real bytes come from a service we cannot call + /// from here, so the test has to construct the same structure and prove the + /// unwrap recovers what went in. + fn envelope( + recipient: &RecipientKey, + payload: &[u8], + content_type: ObjectIdentifier, + key_alg: ObjectIdentifier, + content_alg: ObjectIdentifier, + recipients: usize, + ) -> Vec { + let cek = [0x5au8; 32]; + let iv = [0x11u8; AES_CBC_IV_LEN]; + + // Content, under the content-encryption key. `less_safe_encrypt` pads + // and grows the buffer in place, so it ends up PKCS#7-padded exactly + // as the unwrap expects to find it. + let mut buffer = payload.to_vec(); + let key = UnboundCipherKey::new(&AES_256, &cek).unwrap(); + let encrypting = PaddedBlockEncryptingKey::cbc_pkcs7(key).unwrap(); + encrypting + .less_safe_encrypt(&mut buffer, EncryptionContext::Iv128(iv.into())) + .unwrap(); + + // The content-encryption key, to our public key. + let public = PublicEncryptingKey::from_der(recipient.public_key_der()).unwrap(); + let oaep = OaepPublicEncryptingKey::new(public).unwrap(); + let mut wrapped = vec![0u8; oaep.key_size_bytes()]; + let wrapped_len = oaep + .encrypt(&OAEP_SHA256_MGF1SHA256, &cek, &mut wrapped, None) + .unwrap() + .len(); + wrapped.truncate(wrapped_len); + + let ktri = KeyTransRecipientInfo { + version: cms::content_info::CmsVersion::V0, + rid: RecipientIdentifier::SubjectKeyIdentifier( + x509_cert::ext::pkix::SubjectKeyIdentifier( + OctetString::new(vec![0u8; 20]).unwrap(), + ), + ), + key_enc_alg: AlgorithmIdentifierOwned { + oid: key_alg, + parameters: None, + }, + enc_key: OctetString::new(wrapped).unwrap(), + }; + let mut infos = SetOfVec::new(); + for _ in 0..recipients { + // `SetOfVec` rejects duplicates, so a second recipient differs in + // its identifier — enough to make the count what the test needs. + let mut other = ktri.clone(); + other.rid = RecipientIdentifier::SubjectKeyIdentifier( + x509_cert::ext::pkix::SubjectKeyIdentifier( + OctetString::new(vec![infos.len() as u8; 20]).unwrap(), + ), + ); + infos.insert(RecipientInfo::Ktri(other)).unwrap(); + } + + let enveloped = EnvelopedData { + version: cms::content_info::CmsVersion::V0, + originator_info: None, + recip_infos: RecipientInfos(infos), + encrypted_content: EncryptedContentInfo { + content_type, + content_enc_alg: AlgorithmIdentifierOwned { + oid: content_alg, + parameters: Some( + Any::from_der(&OctetString::new(iv.to_vec()).unwrap().to_der().unwrap()) + .unwrap(), + ), + }, + encrypted_content: Some(OctetString::new(buffer).unwrap()), + }, + unprotected_attrs: None, + }; + + ContentInfo { + content_type: ID_ENVELOPED_DATA, + content: Any::from_der(&enveloped.to_der().unwrap()).unwrap(), + } + .to_der() + .unwrap() + } + + fn well_formed(recipient: &RecipientKey, payload: &[u8]) -> Vec { + envelope( + recipient, + payload, + ID_DATA, + ID_RSAES_OAEP, + ID_AES_256_CBC, + 1, + ) + } + + #[test] + fn a_public_key_is_an_spki_of_a_workable_size() { + let key = RecipientKey::generate().unwrap(); + // The NSM caps `public_key`; an RSA-2048 SPKI is nowhere near it, and + // this is the assertion that would catch a key size change that is. + assert!( + (256..1024).contains(&key.public_key_der().len()), + "unexpected SPKI length {}", + key.public_key_der().len() + ); + PublicEncryptingKey::from_der(key.public_key_der()).expect("a parseable SPKI"); + } + + /// The load-bearing one: what KMS would send comes back out intact. + #[test] + fn a_well_formed_envelope_round_trips() { + let key = RecipientKey::generate().unwrap(); + let secret = [0x42u8; 32]; + let opened = key.unwrap_ciphertext(&well_formed(&key, &secret)).unwrap(); + assert_eq!(opened, secret); + } + + /// A payload that is not a whole number of blocks, to prove the PKCS#7 + /// padding is actually being stripped rather than accidentally absent. + #[test] + fn an_unaligned_payload_round_trips() { + let key = RecipientKey::generate().unwrap(); + let payload = b"thirty-one bytes of content xxx"; + assert_eq!(payload.len() % 16, 15); + let opened = key.unwrap_ciphertext(&well_formed(&key, payload)).unwrap(); + assert_eq!(opened, payload); + } + + /// Two recipients means somebody besides this enclave can open the content. + /// That is the exact thing `Recipient` exists to prevent, so it is refused + /// rather than quietly using the first one. + #[test] + fn more_than_one_recipient_is_refused() { + let key = RecipientKey::generate().unwrap(); + let bytes = envelope(&key, b"secret", ID_DATA, ID_RSAES_OAEP, ID_AES_256_CBC, 2); + let err = key.unwrap_ciphertext(&bytes).unwrap_err().to_string(); + assert!(err.contains("exactly one"), "unexpected error: {err}"); + } + + #[test] + fn an_unexpected_key_transport_algorithm_is_refused() { + let key = RecipientKey::generate().unwrap(); + // rsaEncryption — PKCS#1 v1.5, not OAEP. + let pkcs1 = ObjectIdentifier::new_unwrap("1.2.840.113549.1.1.1"); + let bytes = envelope(&key, b"secret", ID_DATA, pkcs1, ID_AES_256_CBC, 1); + let err = key.unwrap_ciphertext(&bytes).unwrap_err().to_string(); + assert!(err.contains("RSAES-OAEP"), "unexpected error: {err}"); + } + + #[test] + fn an_unexpected_content_encryption_algorithm_is_refused() { + let key = RecipientKey::generate().unwrap(); + // aes128-CBC: the right shape, the wrong strength. + let aes128 = ObjectIdentifier::new_unwrap("2.16.840.1.101.3.4.1.2"); + let bytes = envelope(&key, b"secret", ID_DATA, ID_RSAES_OAEP, aes128, 1); + let err = key.unwrap_ciphertext(&bytes).unwrap_err().to_string(); + assert!(err.contains("AES-256-CBC"), "unexpected error: {err}"); + } + + #[test] + fn an_unexpected_inner_content_type_is_refused() { + let key = RecipientKey::generate().unwrap(); + let bytes = envelope( + &key, + b"secret", + ID_ENVELOPED_DATA, + ID_RSAES_OAEP, + ID_AES_256_CBC, + 1, + ); + let err = key.unwrap_ciphertext(&bytes).unwrap_err().to_string(); + assert!(err.contains("id-data"), "unexpected error: {err}"); + } + + /// An envelope addressed to a different enclave's key. This is the replay + /// case: a document naming a public key we do not hold is useless. + #[test] + fn an_envelope_for_another_key_is_refused() { + let ours = RecipientKey::generate().unwrap(); + let theirs = RecipientKey::generate().unwrap(); + let err = ours + .unwrap_ciphertext(&well_formed(&theirs, b"secret")) + .unwrap_err() + .to_string(); + assert!(err.contains("different public key"), "unexpected: {err}"); + } + + #[test] + fn garbage_is_refused_rather_than_panicking() { + let key = RecipientKey::generate().unwrap(); + assert!(key.unwrap_ciphertext(b"").is_err()); + assert!(key.unwrap_ciphertext(&[0xffu8; 64]).is_err()); + } +} diff --git a/runtime/src/keys/static_key.rs b/runtime/src/keys/static_key.rs new file mode 100644 index 0000000..7a8ebaa --- /dev/null +++ b/runtime/src/keys/static_key.rs @@ -0,0 +1,140 @@ +//! The development key source: a secret from configuration, stored in the clear. + +use anyhow::{bail, Context, Result}; +use async_trait::async_trait; +use s3fs_core::MasterSecret; + +use super::{MasterKeySource, SealedKey}; + +/// Marks a blob that is not sealed at all, so that nothing can mistake one for +/// protection, and so a real source can refuse to open one. +const UNSEALED_MAGIC: &[u8] = b"s3fs-UNSEALED-development-key-v1\n"; + +/// A secret supplied by configuration, stored in the clear. +/// +/// The development seam. `mint` uses the configured secret rather than drawing +/// a fresh one, so a test can predict the filesystem it creates; `open` +/// returns whatever the blob holds, so resume genuinely recovers from stored +/// state rather than from configuration that might since have changed. +pub struct StaticKey(MasterSecret); + +impl StaticKey { + pub fn from_hex(hex: &str) -> Result { + MasterSecret::from_hex(hex) + .map(StaticKey) + .map_err(|e| anyhow::anyhow!("master key: {e}")) + } + + pub fn new(secret: MasterSecret) -> Self { + StaticKey(secret) + } +} + +/// Opaque: a `Debug` that printed the secret would leak it into any log line +/// that formats a struct containing one. +impl std::fmt::Debug for StaticKey { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str("StaticKey()") + } +} + +#[async_trait] +impl MasterKeySource for StaticKey { + fn describe(&self) -> &'static str { + "static, UNSEALED (development only)" + } + + async fn mint(&self) -> Result<(MasterSecret, SealedKey)> { + let mut blob = UNSEALED_MAGIC.to_vec(); + blob.extend_from_slice(self.0.expose_secret()); + Ok((self.0.clone(), SealedKey(blob))) + } + + async fn open(&self, sealed: &SealedKey) -> Result { + let bytes = sealed.as_bytes(); + // Refuse anything this source did not write. A blob sealed by KMS + // opened here would either fail confusingly or, worse, be misread as + // key material — so the shape is checked before anything else. + if !bytes.starts_with(UNSEALED_MAGIC) { + bail!( + "this filesystem's key was sealed by something else — a static \ + key source cannot open it. That is the expected outcome of \ + pointing a development build at a real deployment." + ); + } + let raw = &bytes[UNSEALED_MAGIC.len()..]; + let raw: [u8; 32] = raw + .try_into() + .context("unsealed key blob is not 32 bytes after its marker")?; + Ok(MasterSecret::from_bytes(raw)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn key() -> StaticKey { + StaticKey::new(MasterSecret::from_bytes([5u8; 32])) + } + + #[tokio::test] + async fn a_minted_key_opens_back_to_itself() { + let source = key(); + let (secret, sealed) = source.mint().await.unwrap(); + assert_eq!( + source.open(&sealed).await.unwrap().expose_secret(), + secret.expose_secret() + ); + } + + /// Resume recovers from the *blob*, not from configuration. A deployment + /// whose `S3FS_MASTER_KEY` has drifted must still open the filesystem it + /// created, or the drift would be discovered as unreadable data. + #[tokio::test] + async fn opening_uses_the_blob_rather_than_the_configured_secret() { + let (_, sealed) = key().mint().await.unwrap(); + let different = StaticKey::new(MasterSecret::from_bytes([99u8; 32])); + assert_eq!( + different.open(&sealed).await.unwrap().expose_secret(), + &[5u8; 32] + ); + } + + /// The marker exists so a development build cannot quietly mishandle a + /// real deployment's key blob. + #[tokio::test] + async fn a_blob_from_another_source_is_refused() { + let err = key() + .open(&SealedKey::from_bytes(b"KMS-shaped ciphertext".to_vec())) + .await + .unwrap_err(); + assert!( + format!("{err:#}").contains("sealed by something else"), + "{err:#}" + ); + } + + #[tokio::test] + async fn a_truncated_blob_is_refused_rather_than_padded() { + let mut short = UNSEALED_MAGIC.to_vec(); + short.extend_from_slice(&[1u8; 16]); + assert!(key().open(&SealedKey::from_bytes(short)).await.is_err()); + } + + /// A `Debug` that rendered the blob would print a master key, because for + /// this source the blob *is* the key. + #[test] + fn the_sealed_form_never_debug_prints_its_contents() { + let blob = SealedKey::from_bytes(b"a very secret value".to_vec()); + let rendered = format!("{blob:?}"); + assert!(!rendered.contains("secret value"), "{rendered}"); + assert!(rendered.contains("sha256"), "{rendered}"); + } + + #[test] + fn the_commitment_is_over_the_ciphertext() { + let blob = SealedKey::from_bytes(b"abc".to_vec()); + assert_eq!(blob.sha256(), nitro_attestation::sha256(b"abc")); + } +} diff --git a/runtime/src/lib.rs b/runtime/src/lib.rs new file mode 100644 index 0000000..d70bf0a --- /dev/null +++ b/runtime/src/lib.rs @@ -0,0 +1,100 @@ +//! Everything the enclave runtime does, except parse arguments. +//! +//! A library beside the binary rather than inside it, because integration +//! tests cannot import a binary-only crate — and `tests/serve_tls.rs`, where a +//! client checks that the attestation document binds the certificate from its +//! own handshake, is the test this whole design exists to pass. +//! +//! This was `s3fs-host` until the split stopped describing anything. It began +//! as `wasi:filesystem` over a block store, reusable by any host; it ended up +//! also holding the NSM, the vsock tap device, TLS termination and the +//! attestation endpoints, none of which mean anything outside an enclave. The +//! layer that is genuinely independent is [`s3fs_core`], which still has no +//! wasmtime dependency at all. +//! +//! ```text +//! mount config ──▶ Arc two backends, verified root +//! guest object or file ──▶ PCR16, locked before any key is asked for +//! net vsock + tap ──▶ a network gvforwarder against gvproxy +//! env GuestEnvPolicy ──▶ Vec<(K, V)> inherit minus a denylist +//! linker wasmtime-wasi ──▶ Linker everything but filesystem +//! run component ──▶ GuestOutcome once, with its exit code +//! serve component ──▶ TLS + HTTP until stopped +//! ``` +//! +//! [`wasi`] holds the `wasi:filesystem@0.2.x` implementation. It is written +//! against `s3fs_core::Fs` and knows nothing about AWS or enclaves, so a host +//! with its own state type can still take [`add_filesystem_to_linker`] alone — +//! it is simply no longer packaged as though anyone does. +//! +//! The [`keys`] seam exists so that replacing a configured secret with a KMS +//! release gated on an NSM attestation document is a new implementation of one +//! trait, not a change to any of the above. + +pub mod auth; +pub mod boot; +pub mod clock; +pub mod env; +pub mod flag; +pub mod guest; +pub mod guest_io; +pub mod keys; +pub mod linker; +pub mod mount; +pub mod net; +pub mod notify; +pub mod random; +pub mod run; +pub mod serve; +pub mod state; +pub mod stream; +pub mod tasks; +pub mod tenant; +pub mod wasi; + +#[cfg(any(test, feature = "testing"))] +pub mod testing; + +#[cfg(any(test, feature = "testing"))] +pub use auth::SoftwareAuthenticator; +pub use auth::{ + build_relying_party, AuthEndpoints, ChallengeStore, FilesystemCredentials, Gate, + InteractionScope, TokenStore, AUTH_PREFIX, DEFAULT_CAPACITY, DEFAULT_TOKEN_CAPACITY, +}; +pub use boot::{boot, BootConfig, BootMode, Booted, Pair, ReceiptTrust}; +pub use clock::{open_clock, ClockSource, HostClock, PtpClock, TrustedClock, DEFAULT_PTP_DEVICE}; +pub use env::GuestEnvPolicy; +pub use flag::parse_bool_flag; +pub use guest::{fetch_guest, measure_guest, GuestSource, MAX_GUEST_BYTES}; +pub use guest_io::cloudwatch::{ + boot_marker as guest_log_boot_marker, open_stream as open_guest_log_stream, + start as start_guest_log_forwarder, CloudWatchConfig, CloudWatchDestination, CloudWatchLogSink, + LogDestination, LogForwarder, PutError, PutOutcome, STARTUP_PROBE_TIMEOUT, +}; +pub use guest_io::{ + FanOutSink, GuestLogCollector, GuestLogRecord, GuestLogSink, GuestLogs, GuestStream, + TracingLogSink, GUEST_LOG_TARGET, +}; +pub use keys::{ + open_key_source, KeyPointer, KmsAttestedKey, KmsKeyConfig, MasterKeyConfig, MasterKeySource, + MasterKeySourceKind, SealedKey, StaticKey, +}; +pub use linker::build_linker; +pub use mount::{connect, create, mount_existing, parse_fs_id, Backends, MountConfig, Mounted}; +pub use net::{bring_up, Network, NetworkConfig, NetworkMode, DEFAULT_GVFORWARDER}; +pub use nitro_nsm::{Nsm, NsmDevice, DEFAULT_NSM_DEVICE}; +pub use notify::{ + start as start_notify_forwarder, DeviceRegistry, FcmClient, FcmTransport, Notifier, + NotifyConfig, NotifyContext, NotifyForwarder, SendError, ServiceAccount, + NOTIFY_STARTUP_PROBE_TIMEOUT, +}; +pub use random::{open_entropy, GuestRandom, HostEntropy, RandomSource}; +pub use run::{read_component, GuestEnvironment, GuestOutcome, EXIT_RUNTIME_FAILURE}; +pub use serve::{ + apply_tenant, serve_component, AcmeConfig, CertificateSlot, EgressPolicy, GuestInstance, + PoolLimits, SealedAcmeCache, ServeConfig, ServeHandle, Server, Tenancy, TenantPool, + TlsIdentity, TlsMode, X_ENCLAVE_TENANT, +}; +pub use state::State; +pub use tenant::{tenant_root_by_id, tenants, Arrival, TenantRoot}; +pub use wasi::{add_filesystem_to_linker, S3FsCtxView, S3WasiView}; diff --git a/runtime/src/linker.rs b/runtime/src/linker.rs new file mode 100644 index 0000000..18c3912 --- /dev/null +++ b/runtime/src/linker.rs @@ -0,0 +1,117 @@ +//! WASI linker wiring: everything `wasmtime-wasi` provides *except* +//! `wasi:filesystem`, which [`crate::wasi`] takes over. +//! +//! ## This is a maintenance hazard, deliberately isolated +//! +//! [`add_wasi_minus_filesystem`] is a copy of +//! `wasmtime_wasi::p2::add_to_linker_with_options_async` with the two +//! `filesystem::*` lines removed. Upstream offers no "everything but one +//! interface" entry point, so there is no way to express this except by +//! restating the list. +//! +//! The failure mode is quiet: a `wasmtime-wasi` bump that adds an interface +//! leaves it missing here, and the only symptom is a guest that fails to +//! instantiate with an unresolved import — at run time, in whatever +//! environment happens to try it first. `wasmtime::component::Linker` exposes +//! no way to enumerate what has been registered, so no unit test can prove +//! this list is complete. +//! +//! What mitigates it is that `wasmtime-wasi` is pinned to an exact version, so +//! the list cannot change under us without someone editing a manifest. If you +//! bump it, diff this function against the upstream one. +//! +//! There used to be a second mitigation — CI ran a `wasi:cli/command` guest +//! that imported the full command world, so a missing interface failed that +//! job. It went with the command runner. The runtime serves `wasi:http/proxy` +//! and nothing else now, so the full `wasi:cli` surface is registered because +//! a proxy guest may still reach for parts of it, not because anything +//! requires all of it. + +use wasmtime::component::{HasData, Linker, ResourceTable}; +use wasmtime::Result; +use wasmtime_wasi::cli::{WasiCli, WasiCliView}; +use wasmtime_wasi::clocks::{WasiClocks, WasiClocksView}; +use wasmtime_wasi::random::{WasiRandom, WasiRandomView}; +use wasmtime_wasi::sockets::{WasiSockets, WasiSocketsView}; +use wasmtime_wasi::WasiView; + +use crate::state::State; + +/// Marker for `wasi:io` interfaces, which need `&mut ResourceTable`. +struct HasIo; +impl HasData for HasIo { + type Data<'a> = &'a mut ResourceTable; +} + +/// Add every `wasmtime-wasi` interface except `wasi:filesystem`. +pub fn add_wasi_minus_filesystem(linker: &mut Linker) -> Result<()> { + use wasmtime_wasi::p2::bindings::{cli, clocks, random, sockets}; + use wasmtime_wasi_io::bindings::wasi::io; + let l = linker; + let options = wasmtime_wasi::p2::bindings::LinkOptions::default(); + + // wasi:io (async) + io::error::add_to_linker::(l, |t| t.ctx().table)?; + io::poll::add_to_linker::(l, |t| t.ctx().table)?; + io::streams::add_to_linker::(l, |t| t.ctx().table)?; + + // sockets (async) + sockets::tcp::add_to_linker::(l, State::sockets)?; + sockets::udp::add_to_linker::(l, State::sockets)?; + + // clocks + clocks::wall_clock::add_to_linker::(l, State::clocks)?; + clocks::monotonic_clock::add_to_linker::(l, State::clocks)?; + + // random + random::random::add_to_linker::(l, State::random)?; + random::insecure::add_to_linker::(l, State::random)?; + random::insecure_seed::add_to_linker::(l, State::random)?; + + // cli + cli::exit::add_to_linker::(l, &(&options).into(), State::cli)?; + cli::environment::add_to_linker::(l, State::cli)?; + cli::stdin::add_to_linker::(l, State::cli)?; + cli::stdout::add_to_linker::(l, State::cli)?; + cli::stderr::add_to_linker::(l, State::cli)?; + cli::terminal_input::add_to_linker::(l, State::cli)?; + cli::terminal_output::add_to_linker::(l, State::cli)?; + cli::terminal_stdin::add_to_linker::(l, State::cli)?; + cli::terminal_stdout::add_to_linker::(l, State::cli)?; + cli::terminal_stderr::add_to_linker::(l, State::cli)?; + + // sockets (non-async) + sockets::tcp_create_socket::add_to_linker::(l, State::sockets)?; + sockets::udp_create_socket::add_to_linker::(l, State::sockets)?; + sockets::instance_network::add_to_linker::(l, State::sockets)?; + sockets::network::add_to_linker::(l, &(&options).into(), State::sockets)?; + sockets::ip_name_lookup::add_to_linker::(l, State::sockets)?; + + Ok(()) +} + +/// The full linker a guest sees: WASI, plus `wasi:filesystem` backed by the +/// block store. +pub fn build_linker(engine: &wasmtime::Engine) -> Result> { + let mut linker: Linker = Linker::new(engine); + add_wasi_minus_filesystem(&mut linker)?; + crate::wasi::add_filesystem_to_linker(&mut linker)?; + crate::tasks::add_to_linker(&mut linker)?; + crate::stream::add_to_linker(&mut linker)?; + crate::notify::add_to_linker(&mut linker)?; + Ok(linker) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Not a completeness check — nothing can be, from here. It only catches + /// the coarse failure where two interfaces collide or the filesystem + /// override conflicts with a leftover `wasmtime-wasi` registration. + #[test] + fn the_linker_builds_without_duplicate_registrations() { + let engine = wasmtime::Engine::default(); + build_linker(&engine).expect("linker must build"); + } +} diff --git a/runtime/src/main.rs b/runtime/src/main.rs new file mode 100644 index 0000000..5d877ff --- /dev/null +++ b/runtime/src/main.rs @@ -0,0 +1,1945 @@ +//! `enclave-runtime` — run a Wasm guest against the Merkle-anchored +//! filesystem, inside an AWS Nitro Enclave. +//! +//! An enclave has no shell, no config file, and no operator to type flags at +//! it. It gets whatever the enclave image baked in, so **every setting reads +//! from an environment variable**, with a matching flag that wins if present. +//! `ENV` lines in the image's Dockerfile are the deployment configuration; the +//! flags exist so the same binary stays drivable by hand while developing. +//! +//! The guest component is **not** in the image. The runtime fetches it at boot +//! from `--guest-object`, a key in the roots bucket, then measures it into PCR16 +//! and locks that register before it asks KMS for anything — see +//! [`enclave_runtime::guest`]. PCR0 covers the runtime and *where* the guest +//! comes from; PCR16 covers *what* arrived. A key policy pinning both releases +//! the filesystem key to exactly this runtime running exactly this guest, so an +//! attestation still says which code will read the data — and changing the +//! guest is a policy edit rather than an image rebuild. +//! +//! Argument parsing over [`enclave_runtime`], which is the same crate: the +//! library half is everything below this file, and lives beside it rather than +//! inside it so integration tests can reach it. +//! +//! The master secret comes from `--master-key-source`, which has no default: +//! `kms` mints it inside the enclave and lets KMS release it only against an +//! attestation whose PCR0 and PCR16 match the key policy, and `static` takes it from +//! configuration for development and for the QEMU harness. Supplying a +//! plaintext key under `kms` is refused rather than ignored — a key in an +//! environment variable is visible to the parent instance, exactly the party +//! an enclave exists to exclude. + +use std::net::SocketAddr; +use std::path::PathBuf; +use std::time::Duration; + +use anyhow::{Context as _, Result}; +use clap::Parser; +use enclave_runtime::{ + open_clock, open_entropy, serve_component, AcmeConfig, ClockSource, GuestEnvPolicy, + GuestEnvironment, GuestSource, MasterKeySource, MountConfig, NetworkConfig, NetworkMode, + RandomSource, ReceiptTrust, ServeConfig, TlsMode, DEFAULT_GVFORWARDER, DEFAULT_NSM_DEVICE, + DEFAULT_PTP_DEVICE, EXIT_RUNTIME_FAILURE, +}; + +/// Where the FCM credential comes from, once the settings have been checked. +enum FcmCredential { + /// Already parsed, because it was supplied literally and parsing it needed + /// nothing but the bytes. + Account(Box), + /// A name to resolve against SSM, once there is a network. + Parameter(String), +} + +/// Notification settings, as far as they can be decided without I/O. +struct NotifySettings { + project_id: String, + credential: FcmCredential, + endpoint: Option, +} + +/// Read a parameter, decrypting a SecureString if that is what it is. +/// +/// Its own small client rather than the one `keys` builds: that one is +/// configured from a `MasterKeyConfig` and exists to fetch a KMS ciphertext, +/// and threading an unrelated credential through it would tie two things +/// together that have no reason to change at the same time. +async fn read_ssm_parameter(cli: &Cli, name: &str) -> Result { + use aws_sdk_ssm::config::{BehaviorVersion, Credentials, Region}; + + let region = Region::new(cli.region.clone()); + let mut builder = aws_sdk_ssm::Config::builder() + .behavior_version(BehaviorVersion::latest()) + .region(region.clone()) + // Bounded, because this runs before the listener binds. A parameter + // store that accepts a connection and then says nothing would mean an + // enclave that never serves at all. + .timeout_config( + aws_sdk_ssm::config::timeout::TimeoutConfig::builder() + .operation_timeout(Duration::from_secs(5)) + .connect_timeout(Duration::from_secs(3)) + .build(), + ); + if let Some(endpoint) = &cli.ssm_endpoint { + builder = builder.endpoint_url(endpoint); + } + // Static credentials when they were configured, and the default chain + // otherwise — which inside an enclave reaches the parent's instance role + // through gvproxy. Naming one is not optional: `Config::builder()` starts + // empty and resolves neither on its own, so leaving this out is not a + // default, it is no credentials at all. + builder = match ( + cli.access_key_id.as_deref(), + cli.secret_access_key.as_deref(), + ) { + (Some(akid), Some(sak)) => builder.credentials_provider(Credentials::new( + akid, + sak, + cli.session_token.clone(), + None, + "s3fs-static", + )), + _ => builder.credentials_provider( + aws_config::default_provider::credentials::DefaultCredentialsChain::builder() + .region(region) + .build() + .await, + ), + }; + let client = aws_sdk_ssm::Client::from_conf(builder.build()); + let response = client + .get_parameter() + .name(name) + .with_decryption(true) + .send() + .await + .with_context(|| format!("reading SSM parameter {name}"))?; + response + .parameter() + .and_then(|p| p.value()) + .map(str::to_string) + .with_context(|| format!("SSM parameter {name} has no value")) +} + +#[derive(Parser, Debug)] +#[command( + version, + about = "Run a Wasm guest against a Merkle-anchored S3 filesystem", + long_about = None +)] +struct Cli { + /// Enable durable background tasks. Requires authentication and a guest + /// exporting run-task from enclave:tasks/background@0.1.0. + #[arg(long, env = "S3FS_BACKGROUND_TASKS", default_value = "false", value_parser = enclave_runtime::parse_bool_flag, action = clap::ArgAction::Set)] + background_tasks: bool, + #[arg(long, env = "S3FS_BACKGROUND_CONCURRENCY", default_value_t = 1)] + background_concurrency: usize, + #[arg(long, env = "S3FS_BACKGROUND_TIMEOUT_SECS", default_value_t = 30)] + background_timeout_secs: u64, + #[arg(long, env = "S3FS_BACKGROUND_MAX_RECORDS", default_value_t = 1024)] + background_max_records: usize, + #[arg(long, env = "S3FS_BACKGROUND_PER_TENANT", default_value_t = 64)] + background_per_tenant: usize, + + /// Guest component to run, as a key in the roots bucket. What an enclave + /// image uses. + /// + /// Fetched at boot, measured into PCR16 and locked before any key is asked + /// for. The key is baked into the image, so PCR0 covers where the guest + /// comes from; PCR16 covers what arrived. The object itself need not be + /// trusted: a substituted one measures to a different PCR16, which a key + /// policy pinning the approved guest releases nothing to. + #[arg(long, env = "S3FS_GUEST_OBJECT")] + guest_object: Option, + + /// Guest component to run, from a local file. For development and tests, + /// and measured exactly as an object is. Give this or `--guest-object`. + #[arg(long, env = "S3FS_GUEST_PATH")] + guest_path: Option, + + /// Bucket holding the data slabs. + #[arg(long, env = "S3FS_BUCKET")] + bucket: String, + + /// Bucket holding the signed root records. Defaults to `--bucket`. + /// + /// These should differ in production: the roots bucket carries Object Lock + /// COMPLIANCE retention and is the entire rollback guarantee, while the + /// data bucket stays unlocked so dead copy-on-write blocks stay + /// reclaimable. + #[arg(long, env = "S3FS_ROOTS_BUCKET")] + roots_bucket: Option, + + /// Where the master secret comes from. + /// + /// `kms` mints it inside the enclave and lets KMS release it only against + /// an attestation whose PCR0 and PCR16 match the key policy — so a wrong + /// image or a wrong guest gets no key at all, rather than a refused mount. + /// `static` takes it from `--master-key` and stores it unsealed; it exists + /// for development and for the QEMU harness, whose emulated NSM cannot + /// produce a document KMS would accept. + /// + /// No default. This is the one setting that should never be inherited by + /// omission, and because the image environment is measured, PCR0 records + /// which of the two an enclave was built with. + #[arg(long, env = "S3FS_MASTER_KEY_SOURCE", + value_parser = enclave_runtime::MasterKeySourceKind::parse)] + master_key_source: enclave_runtime::MasterKeySourceKind, + + /// 32-byte master secret, hex encoded. **`--master-key-source=static` only.** + /// + /// Supplying it under `kms` is refused rather than ignored: a key from + /// configuration is a key the parent instance holds, which is precisely + /// what KMS release exists to prevent. + #[arg(long, env = "S3FS_MASTER_KEY")] + master_key: Option, + + /// The customer master key that releases this filesystem's secret. Its + /// policy — `kms:RecipientAttestation:PCR0` and `:PCR16`, on both + /// `GenerateDataKey` and `Decrypt` — is the security control. + #[arg(long, env = "S3FS_KMS_KEY_ID")] + kms_key_id: Option, + + /// SSM parameter holding the KMS ciphertext, and nothing else. + #[arg(long, env = "S3FS_MASTER_KEY_PARAMETER")] + master_key_parameter: Option, + + /// Deployment name, mixed into the KMS encryption context alongside the + /// filesystem id — so a staging enclave cannot open a production + /// filesystem even when pointed at the same parameter. + #[arg(long, env = "S3FS_ENVIRONMENT", default_value = "production")] + environment: String, + + /// Endpoint overrides for KMS and SSM, separate from `--endpoint`. + /// + /// S3's override points at MinIO in development; a MinIO endpoint is not a + /// KMS endpoint, and reusing it would fail in a way that reads like a + /// credentials problem. + #[arg(long, env = "S3FS_KMS_ENDPOINT")] + kms_endpoint: Option, + + #[arg(long, env = "S3FS_SSM_ENDPOINT")] + ssm_endpoint: Option, + + /// Filesystem identifier, 32 hex characters — the key-derivation salt, so + /// two filesystems under one master secret stay independent. + /// + /// Supplied rather than read from the store: the keys that verify a root + /// record derive from it, so taking it from the store would mean trusting + /// the store to say which key checks its own signature. + #[arg( + long, + env = "S3FS_ID", + default_value = "00000000000000000000000000000000" + )] + fs_id: String, + + /// Refuse to mount a root record older than this sequence number. + /// + /// The only defence against a store that hides newer roots at a cold + /// mount. Everything else about rollback is closed cryptographically; this + /// one needs a number from outside the store. + #[arg(long, env = "S3FS_MIN_ROOT_SEQ")] + min_root_seq: Option, + + /// Path the guest sees as its preopen root. + #[arg(long, env = "S3FS_MOUNT_PATH", default_value = "/")] + mount_path: String, + + /// Key prefix inside both buckets. + #[arg(long, env = "S3FS_BUCKET_PREFIX", default_value = "")] + bucket_prefix: String, + + #[arg(long, env = "AWS_REGION", default_value = "us-east-1")] + region: String, + + /// Endpoint override, e.g. a local MinIO or the parent's vsock proxy. + #[arg(long, env = "S3FS_ENDPOINT")] + endpoint: Option, + + #[arg(long, env = "AWS_ACCESS_KEY_ID")] + access_key_id: Option, + + #[arg(long, env = "AWS_SECRET_ACCESS_KEY")] + secret_access_key: Option, + + #[arg(long, env = "AWS_SESSION_TOKEN")] + session_token: Option, + + /// Path-style addressing, required by MinIO and many S3-compatibles. + #[arg(long, env = "S3FS_FORCE_PATH_STYLE", value_parser = enclave_runtime::parse_bool_flag, num_args = 0..=1, default_value_t = false, default_missing_value = "true")] + force_path_style: bool, + + /// Skip the `HeadBucket` startup probe. + #[arg(long, env = "S3FS_SKIP_BUCKET_PROBE", value_parser = enclave_runtime::parse_bool_flag, num_args = 0..=1, default_value_t = false, default_missing_value = "true")] + skip_bucket_probe: bool, + + /// Give the guest nothing but what `--guest-env` names. + /// + /// **Use this when running outside an enclave.** Inheritance is right for + /// an enclave image, where the environment is curated deployment + /// configuration; it is wrong on a developer's machine, where it forwards + /// that machine's whole environment to the guest. The denylist below + /// withholds this runtime's own credentials, not a `GITHUB_TOKEN` or an + /// `SSH_AUTH_SOCK`. + /// + /// By default it inherits this process's environment minus anything under + /// `AWS_` or `S3FS_`, which is where the credentials and this runtime's + /// own configuration live. + #[arg(long, env = "S3FS_NO_INHERIT_ENV", value_parser = enclave_runtime::parse_bool_flag, num_args = 0..=1, default_value_t = false, default_missing_value = "true")] + no_inherit_env: bool, + + /// Extra variable for the guest, as `NAME` (inherit that one by name) or + /// `NAME=VALUE`. Applied after inheritance, so it overrides — including + /// for names that are otherwise withheld. Repeatable, or comma-separated. + /// + /// Paired with `--no-inherit-env` this is the explicit model: the guest + /// gets exactly what is named here and nothing else. + #[arg( + long = "guest-env", + env = "S3FS_GUEST_ENV", + value_delimiter = ',', + value_name = "NAME[=VALUE]" + )] + guest_env: Vec, + + /// Where the guest's wall-clock time comes from. + /// + /// `ptp` reads the Nitro PTP hardware clock and refuses to start without + /// it. `host` uses the system clock, which inside an enclave is whatever + /// the hypervisor last set — fine for development, untrusted in + /// production. `auto` prefers PTP and warns when it falls back. + /// + /// An enclave image should set this to `ptp`. + #[arg(long, env = "S3FS_CLOCK_SOURCE", default_value = "auto", + value_parser = ClockSource::parse)] + clock_source: ClockSource, + + /// PTP character device to read. + #[arg(long, env = "S3FS_PTP_DEVICE", default_value = DEFAULT_PTP_DEVICE)] + ptp_device: PathBuf, + + /// Where the guest's random bytes come from. + /// + /// `nsm` reads the Nitro Security Module and refuses to start without it. + /// `host` uses the kernel, which inside an enclave is NSM-seeded but says + /// nothing about it. `auto` prefers NSM and reports a fallback as an error, + /// because every key the guest generates afterwards rests on the answer. + /// + /// An enclave image should set this to `nsm`. + #[arg(long, env = "S3FS_RANDOM_SOURCE", default_value = "auto", + value_parser = RandomSource::parse)] + random_source: RandomSource, + + /// NSM character device to read. + #[arg(long, env = "S3FS_NSM_DEVICE", default_value = DEFAULT_NSM_DEVICE)] + nsm_device: PathBuf, + + /// How a state-origin receipt must be trusted. + /// + /// `required` demands a signature chaining to the AWS Nitro root. + /// `unsigned-emulator` reads the contents without checking anything, + /// because QEMU's emulated NSM does not sign — and because the image's + /// environment is measured, PCR0 tells a client which of the two an + /// enclave was built with. A production image never sets it. + #[arg(long, env = "S3FS_RECEIPT_TRUST", default_value = "required", + value_parser = ReceiptTrust::parse)] + receipt_trust: ReceiptTrust, + + /// Seconds a guest may take to produce a response head before the request + /// is abandoned. + /// + /// A guest that neither returns nor answers otherwise hangs forever, and + /// with one request in flight at a time that is the whole server. Baked + /// into the image like every other setting, so PCR0 covers it. + #[arg(long, env = "S3FS_REQUEST_TIMEOUT_SECS", default_value_t = 30)] + request_timeout_secs: u64, + + /// Seconds one S3 request may take, before the SDK's retries. + /// + /// Its own setting rather than a reuse of `--request-timeout-secs`, which + /// bounds how long a *guest* may take: tuning how long a guest may think + /// should not silently change how long a slab read may take, and the two + /// have no reason to move together. + /// + /// It has to be a setting at all because an enclave has no shell. Thirty + /// seconds is a guess, and a slow endpoint or a large slab is exactly the + /// case where a guess is wrong — with the symptom being a mount that never + /// returns rather than an error that names S3. + #[arg(long, env = "S3FS_S3_TIMEOUT_SECS", default_value_t = 30)] + s3_timeout_secs: u64, + + #[arg(long, env = "S3FS_GUEST_LOG_GROUP")] + guest_log_group: Option, + + /// The stream within `--guest-log-group`. Required alongside it. + #[arg(long, env = "S3FS_GUEST_LOG_STREAM")] + guest_log_stream: Option, + + /// Point the CloudWatch client somewhere else. For tests. + /// + /// A downgrade path, and worth naming as one: aimed at an `http://` + /// endpoint it hands guest output to whatever is listening, in clear. That + /// is tolerable only because this is baked into the image, so PCR0 records + /// which was built. + #[arg(long, env = "S3FS_GUEST_LOG_ENDPOINT")] + guest_log_endpoint: Option, + + /// The Firebase project wake signals are sent for. + /// + /// Empty means notifications are off — the same layering rule the guest + /// log settings follow, so an image can carry the setting and a deployment + /// can blank it. + #[arg(long, env = "S3FS_FCM_PROJECT_ID")] + fcm_project_id: Option, + + /// The service-account JSON itself. Development and local runs. + /// + /// Inside an enclave this arrives through the parent instance, which is the + /// party the enclave excludes. What a stolen credential buys is the ability + /// to ring doorbells: wake signals carry no data, and reading any of it + /// still needs a key KMS releases only to a matching PCR0/PCR16. + #[arg(long, env = "S3FS_FCM_SERVICE_ACCOUNT")] + fcm_service_account: Option, + + /// An SSM parameter holding that JSON. The production source. + /// + /// Keeps the secret out of the measured image and out of the launch + /// invocation, and lets it rotate without moving PCR0. + #[arg(long, env = "S3FS_FCM_SERVICE_ACCOUNT_PARAMETER")] + fcm_service_account_parameter: Option, + + /// Point the FCM client somewhere else. For tests and the emulator. + /// + /// A downgrade path: an `http://` endpoint hands wake signals to whatever + /// is listening. PCR0 records which image was built, which is the only + /// reason this is acceptable. + #[arg(long, env = "S3FS_FCM_ENDPOINT")] + fcm_endpoint: Option, + + /// Clients kept warm at once. + /// + /// A warm client costs a wasm linear memory and nothing else — the + /// filesystem and its block cache are shared — so this bounds instance + /// memory rather than cache memory. Past it, the least recently used idle + /// client is dropped; one serving a request is never evicted. + #[arg(long, env = "S3FS_MAX_TENANTS", default_value_t = 64)] + max_tenants: usize, + + /// Seconds a mounted client may sit idle before it is dropped. + #[arg(long, env = "S3FS_TENANT_IDLE_SECS", default_value_t = 900)] + tenant_idle_secs: u64, + + /// Requests one client's instance serves before it is rebuilt. + /// + /// Wasm linear memory never shrinks, so an instance that lived forever + /// would only grow. Rebuilding costs ~24 µs and keeps the mount, which is + /// the part that costs S3 round trips. + #[arg(long, env = "S3FS_MAX_REQUESTS_PER_INSTANCE", default_value_t = 10_000)] + max_requests_per_instance: u64, + + /// The WebAuthn relying-party id — the domain passkeys are scoped to. + /// + /// Setting it, with `--webauthn-origin`, is what turns authentication on: + /// **every request that could reach the guest then needs a fresh assertion + /// bound to exactly that request.** Left unset the runtime serves the guest + /// to anyone who can open a connection, which is a development and QEMU + /// arrangement and is warned about at startup. + /// + /// Must match the domain the app is scoped to, and production needs + /// browser-trusted TLS on it — see `--tls acme`. Baked into the image, so + /// PCR0 records which relying party an enclave will accept assertions for. + #[arg(long, env = "S3FS_WEBAUTHN_RP_ID")] + webauthn_rp_id: Option, + + /// The exact origin assertions must claim, e.g. `https://cosigner.example.com`. + /// + /// Compared exactly, not by suffix: a page on another origin must not be + /// able to borrow a user's passkey for this one. + #[arg(long, env = "S3FS_WEBAUTHN_ORIGIN")] + webauthn_origin: Option, + + /// A further origin assertions may claim. Repeatable. + /// + /// For native apps, which do not claim `https://`. An Android app + /// claims `android:apk-key-hash:`: the unpadded base64url SHA-256 of + /// its signing certificate. Debug, upload and Play App Signing keys each + /// have their own, so a deployment usually lists the one Play signs with. + /// Android lets an app claim it only after the app is vouched for by + /// `https:///.well-known/assetlinks.json`. + /// + /// Compared exactly, like `--webauthn-origin`. Baked into the image, so + /// PCR0 records which apps an enclave accepts assertions from. + #[arg( + long = "webauthn-allowed-origin", + env = "S3FS_WEBAUTHN_ALLOWED_ORIGINS", + value_delimiter = ',' + )] + webauthn_allowed_origins: Vec, + + /// An origin guests may send `wasi:http` requests to, as + /// `https://host[:port]` — or `http://host:port`, for a local stack. + /// Repeatable. None by default, and then a guest has no outbound network at + /// all. + /// + /// Compared exactly: scheme, host and port. It is a channel out of the + /// enclave carrying whatever the guest puts in it, so name only services + /// the guest has to reach. Image environment, so PCR0 covers the list. + #[arg( + long = "guest-egress-origin", + env = "S3FS_GUEST_EGRESS_ORIGINS", + value_delimiter = ',' + )] + guest_egress_origins: Vec, + + /// Seconds a challenge is good for. + /// + /// Long enough for a person to look at a prompt and present a finger, + /// short enough that a captured assertion is stale before it can be used. + #[arg(long, env = "S3FS_CHALLENGE_TTL_SECS", default_value_t = 60)] + challenge_ttl_secs: u64, + + /// Seconds an interaction token is good for before it is spent. + /// + /// This bounds the time to *start* an interaction — a person who approves + /// something and then puts their phone down should not find the approval + /// still live later. It is not how long an interaction may run. + #[arg(long, env = "S3FS_INTERACTION_TOKEN_TTL_SECS", default_value_t = 60)] + interaction_token_ttl_secs: u64, + + /// Seconds an interaction may run once started. + /// + /// The other half of the distinction: a stream holds its tenant's single + /// slot for its whole life, so this is what bounds how long that tenant's + /// next request waits. Enforced by wall clock, because a guest parked in a + /// host call executes no wasm and the epoch cannot see it. + #[arg(long, env = "S3FS_MAX_INTERACTION_SECS", default_value_t = 300)] + max_interaction_secs: u64, + + /// Report on the configured clock and entropy source and exit, without + /// mounting or running anything. For diagnosing a deployment, and the only + /// thing the emulator harness needs — it touches no storage. + #[arg(long, alias = "clock-check")] + self_check: bool, + + /// Address to serve on in `serve` mode. + /// + /// Binds every interface by default because inside an enclave the only + /// interface is the one the parent's proxy reaches, and binding loopback + /// there would answer nobody. + #[arg(long, env = "S3FS_HTTP_LISTEN", default_value = "0.0.0.0:8080")] + http_listen: SocketAddr, + + /// Where the serving certificate comes from. + /// + /// `acme`, the default, obtains one from Let's Encrypt over TLS-ALPN-01 on + /// the same port. It needs `--tls-domain` and outbound network. The key is + /// generated inside the enclave and never leaves it; what the CA signs is + /// what every response's attestation document binds. + /// + /// `off` serves plaintext. Inside an enclave that hands every request to + /// the parent instance, which is the party this design excludes. + /// + /// There is deliberately no self-signed mode at all: a certificate an + /// operator can mint is one they can mint for an impostor too, and it buys + /// nothing the attestation binding does not already give. A deployment with + /// no public CA points `--acme-directory` at a private one instead. + #[arg(long, env = "S3FS_TLS", default_value = "acme", + value_parser = TlsMode::parse)] + tls: TlsMode, + + /// Domain to put in the certificate. Repeatable; required for `--tls acme`. + #[arg(long = "tls-domain", env = "S3FS_TLS_DOMAINS", value_delimiter = ',')] + tls_domains: Vec, + + /// Contact address registered with the ACME provider, for expiry notices. + #[arg( + long = "acme-contact", + env = "S3FS_ACME_CONTACTS", + value_delimiter = ',' + )] + acme_contacts: Vec, + + /// ACME directory URL. Defaults to Let's Encrypt production. + /// + /// Point it at staging while setting a deployment up. Production allows + /// five duplicate certificates per week per domain, and an enclave that + /// keeps re-issuing will exhaust that and be unable to serve. + #[arg(long, env = "S3FS_ACME_DIRECTORY")] + acme_directory: Option, + + /// PEM trust root for the ACME directory's own HTTPS certificate. + /// + /// **Only in a `testing` build.** A real CA serves its API under a + /// publicly-trusted certificate and needs nothing here; a test CA like + /// Pebble does not, so the QEMU harness supplies its root. A production + /// enclave has no such flag, because one that took issuance orders from a + /// CA of the operator's choosing would be a different trust model than the + /// one this image claims. + #[cfg(any(test, feature = "testing"))] + #[arg(long, env = "S3FS_ACME_CA")] + acme_ca: Option, + + /// Re-sign attestation documents with a certificate chain minted at boot. + /// + /// **Only in a `testing` build, and only useful under an emulator.** QEMU's + /// NSM produces documents with genuine contents — the real PCR0 of the + /// image, the PCR16 the runtime measured — inside an envelope it does not + /// sign: its source says *"we don't actually sign the data, so we use -1 as + /// the 'alg' value"*, and -1 is not a COSE algorithm. A client meeting one + /// has to pass `--unsigned-emulator`, which skips the signature, the chain + /// and the validity windows entirely. + /// + /// With this set, the runtime asks the device for a document and re-signs + /// that same payload, so a client pins the reported root and runs every + /// check it would run against hardware. It proves nothing about *who* + /// produced a document — the key is minted inside an image its operator + /// controls — which is why a production binary has no such flag. + #[cfg(any(test, feature = "testing"))] + #[arg(long, env = "S3FS_COSIGN_ATTESTATIONS", value_parser = enclave_runtime::parse_bool_flag, num_args = 0..=1, default_value_t = false, default_missing_value = "true")] + cosign_attestations: bool, + + /// How the enclave reaches the network. + /// + /// `gvproxy` runs the tap forwarder against the parent's gvproxy, which is + /// the only way an enclave gets an interface at all — it has no NIC, only + /// vsock. Without it nothing that speaks TCP works: not S3, not ACME, not + /// even DNS. + /// + /// `none` assumes the network is already there, which is true everywhere + /// except inside an enclave. + /// + /// An enclave image should set this to `gvproxy`. + #[arg(long, env = "S3FS_NETWORK", default_value = "none", + value_parser = NetworkMode::parse)] + network: NetworkMode, + + /// The tap forwarder binary, shipped inside the enclave image. + #[arg(long, env = "S3FS_GVFORWARDER", default_value = DEFAULT_GVFORWARDER)] + gvforwarder: PathBuf, + + /// Arguments passed to the guest. + #[arg(last = true)] + guest_args: Vec, +} + +impl Cli { + /// Where guest output goes, beyond the console. + /// + /// A group without a stream is refused here rather than at the first write: + /// it needs no I/O to know it is wrong, and the same discipline as + /// `--webauthn-rp-id`/`--webauthn-origin`. Everything that *does* need the + /// network is decided at startup instead, where a transient failure can be + /// told apart from a mistake. + /// Everything about notifications that can be decided without a network. + /// + /// The literal credential is parsed here, key and all: a malformed service + /// account must fail at boot rather than the first time somebody is waiting + /// to be woken, and proving it parses needs nothing but the bytes. + fn notify_settings(&self) -> Result> { + let set = |value: &Option| { + value + .as_deref() + .map(str::trim) + .filter(|v| !v.is_empty()) + .map(str::to_string) + }; + let project = set(&self.fcm_project_id); + let literal = set(&self.fcm_service_account); + let parameter = set(&self.fcm_service_account_parameter); + + // Empty means off, not malformed. + if project.is_none() && literal.is_none() && parameter.is_none() { + return Ok(None); + } + anyhow::ensure!( + !(literal.is_some() && parameter.is_some()), + "--fcm-service-account and --fcm-service-account-parameter are alternatives. \ + Refused rather than resolved by precedence: a deployment that set both has \ + one of them wrong, and guessing which would be the wrong help." + ); + let Some(project_id) = project else { + anyhow::bail!( + "an FCM credential was given without --fcm-project-id, so there is no \ + project to send for" + ); + }; + let credential = match (literal, parameter) { + (Some(json), _) => FcmCredential::Account(Box::new( + enclave_runtime::ServiceAccount::parse(&json).context("--fcm-service-account")?, + )), + (_, Some(name)) => FcmCredential::Parameter(name), + (None, None) => anyhow::bail!( + "--fcm-project-id was set without a service account, so nothing can be sent" + ), + }; + Ok(Some(NotifySettings { + project_id, + credential, + endpoint: set(&self.fcm_endpoint), + })) + } + + fn guest_log_config(&self) -> Result> { + // Empty means off, not malformed. The image environment is layered — + // the QEMU image inherits production's and overrides what it cannot + // use — and setting a variable to "" is how that layer says "not this + // one". Treating it as a typo would make the override impossible. + let group = self + .guest_log_group + .as_deref() + .map(str::trim) + .filter(|g| !g.is_empty()); + let stream = self + .guest_log_stream + .as_deref() + .map(str::trim) + .filter(|s| !s.is_empty()); + match (group, stream) { + (None, None) => Ok(None), + (Some(group), Some(stream)) => Ok(Some(enclave_runtime::CloudWatchConfig { + log_group: group.to_string(), + log_stream: stream.to_string(), + region: self.region.clone(), + endpoint: self.guest_log_endpoint.clone(), + })), + _ => anyhow::bail!( + "--guest-log-group and --guest-log-stream must be given together. \ + This runtime holds `logs:PutLogEvents` and nothing more, so it cannot \ + create a stream it was not told the name of." + ), + } + } + + /// Where the guest comes from: exactly one of the two settings. + /// + /// An empty setting counts as unset, for the layering reason + /// `guest_log_config` gives. + fn guest_source(&self) -> Result { + let object = self + .guest_object + .as_deref() + .map(str::trim) + .filter(|key| !key.is_empty()); + let path = self + .guest_path + .as_ref() + .filter(|path| !path.as_os_str().is_empty()); + match (object, path) { + (Some(key), None) => Ok(GuestSource::Object { + key: key.to_string(), + }), + (None, Some(path)) => Ok(GuestSource::Path(path.clone())), + (None, None) => anyhow::bail!( + "no guest to run: set --guest-object, a key in the roots bucket, or outside \ + an enclave --guest-path" + ), + (Some(_), Some(_)) => anyhow::bail!( + "--guest-object and --guest-path are both set. They name two guests, and \ + there is no safe way to guess which was meant." + ), + } + } + + fn mount_config(&self) -> Result { + Ok(MountConfig { + bucket: self.bucket.clone(), + roots_bucket: self.roots_bucket.clone(), + region: self.region.clone(), + endpoint: self.endpoint.clone(), + access_key_id: self.access_key_id.clone(), + secret_access_key: self.secret_access_key.clone(), + session_token: self.session_token.clone(), + force_path_style: self.force_path_style, + bucket_prefix: self.bucket_prefix.clone(), + mount_path: self.mount_path.clone(), + fs_id: enclave_runtime::parse_fs_id(&self.fs_id)?, + min_root_seq: self.min_root_seq, + skip_bucket_probe: self.skip_bucket_probe, + request_timeout: Duration::from_secs(self.s3_timeout_secs), + }) + } + + fn master_key_config(&self) -> Result { + Ok(enclave_runtime::MasterKeyConfig { + kind: self.master_key_source, + master_key: self.master_key.clone(), + kms_key_id: self.kms_key_id.clone(), + parameter: self.master_key_parameter.clone(), + environment: self.environment.clone(), + fs_id: enclave_runtime::parse_fs_id(&self.fs_id)?, + region: self.region.clone(), + kms_endpoint: self.kms_endpoint.clone(), + ssm_endpoint: self.ssm_endpoint.clone(), + access_key_id: self.access_key_id.clone(), + secret_access_key: self.secret_access_key.clone(), + session_token: self.session_token.clone(), + }) + } + + fn env_policy(&self) -> GuestEnvPolicy { + if self.no_inherit_env { + GuestEnvPolicy::explicit_only(self.guest_env.clone()) + } else { + GuestEnvPolicy { + inherit: true, + explicit: self.guest_env.clone(), + } + } + } +} + +#[tokio::main] +async fn main() -> std::process::ExitCode { + init_tracing(); + + match run().await { + Ok(outcome) => { + if outcome.is_success() { + tracing::info!("guest exited successfully"); + } else { + tracing::warn!(exit_code = outcome.exit_code(), "guest exited non-zero"); + } + exit_code(outcome.exit_code()) + } + Err(e) => { + tracing::error!( + error = format!("{e:#}"), + "runtime failed to start the guest" + ); + exit_code(EXIT_RUNTIME_FAILURE) + } + } +} + +async fn run() -> Result { + let cli = Cli::parse(); + + let clock = open_clock(cli.clock_source, &cli.ptp_device)?; + let entropy = open_entropy(cli.random_source, &cli.nsm_device)?; + + /// Wrap the device so its documents carry a signature, if this build can. + /// + /// Two definitions rather than a runtime branch, for the reason `--acme-ca` + /// has two: a production binary has no flag to read, so no deployment can + /// talk it into signing attestations with a key it minted itself. + #[cfg(any(test, feature = "testing"))] + fn cosign( + cli: &Cli, + entropy: std::sync::Arc, + ) -> Result> { + use base64::Engine as _; + + if !cli.cosign_attestations { + return Ok(entropy); + } + let cosigning = enclave_runtime::testing::CosigningNsm::wrap(entropy)?; + // At `warn`, and printed in full. A client cannot check anything + // without this value, and it is minted fresh at every boot — so it has + // to leave by the one channel the parent already reads, and it has to + // stand out from the boot log around it. + tracing::warn!( + trust_root = %base64::engine::general_purpose::STANDARD.encode(cosigning.trust_root()), + "attestation documents are signed with a chain minted at boot, not by Nitro \ + hardware; a client must pin the root below, and it means only that this image \ + produced the document" + ); + Ok(std::sync::Arc::new(cosigning)) + } + + #[cfg(not(any(test, feature = "testing")))] + fn cosign( + _cli: &Cli, + entropy: std::sync::Arc, + ) -> Result> { + Ok(entropy) + } + + let entropy = cosign(&cli, entropy)?; + + if cli.self_check { + clock_check(clock.as_ref())?; + println!(); + entropy_check(entropy.as_ref())?; + return Ok(enclave_runtime::GuestOutcome::Success); + } + + // Before anything that needs a socket. Mounting reaches S3, so a runtime + // that mounted first would fail with an S3 error that says nothing about + // the real cause. + let _network = enclave_runtime::bring_up(&NetworkConfig { + mode: cli.network, + gvforwarder: cli.gvforwarder.clone(), + ..Default::default() + })?; + + let keys = enclave_runtime::open_key_source(&cli.master_key_config()?, entropy.clone())?; + // Resolved before anything touches the network, so a runtime given no + // guest, or two, says so rather than failing somewhere later. + let guest_source = cli.guest_source()?; + tracing::info!( + key_source = keys.describe(), + guest = %guest_source, + "starting" + ); + + let mount_config = cli.mount_config()?; + let backends = enclave_runtime::connect(&mount_config).await?; + + // The guest, before anything asks for a key. KMS releases the key against + // an attestation carrying PCR16, so PCR16 has to be final by then: measured, + // and locked so nothing can extend it afterwards. These same bytes are what + // get compiled and served below, so what was measured is what runs. + let component = enclave_runtime::fetch_guest(&guest_source, &backends.roots).await?; + let guest_pcr = enclave_runtime::measure_guest(entropy.as_ref(), &component)?; + tracing::info!( + guest = %guest_source, + guest_sha256 = %hex::encode(nitro_attestation::sha256(&component)), + pcr16 = %hex::encode(guest_pcr), + "guest measured into PCR16 and locked" + ); + + // Decide whether this enclave is entitled to the state it is about to + // load, before it loads any of it. + let booted = enclave_runtime::boot( + &backends, + &mount_config, + &enclave_runtime::BootConfig { + trust: cli.receipt_trust, + }, + &entropy, + &keys, + ) + .await?; + + tracing::info!( + mode = ?booted.mode, + pcr0 = %hex::encode(booted.pair.pcr0), + pcr16 = %hex::encode(booted.pair.pcr16), + state_root = %hex::encode(booted.state_root), + "state origin established" + ); + + let mounted = booted.mounted; + let fs = mounted.fs.clone(); + + let env = cli.env_policy().build()?; + + /// The ACME directory's trust root, if this build can have one. + /// + /// Two definitions rather than a runtime branch: a production binary has no + /// `--acme-ca` field to read, so the public roots are not something a + /// deployment can talk it out of. + #[cfg(any(test, feature = "testing"))] + fn acme_directory_ca(cli: &Cli) -> anyhow::Result>> { + match &cli.acme_ca { + Some(path) => Ok(Some(anyhow::Context::with_context( + std::fs::read(path), + || format!("reading the ACME directory trust root {}", path.display()), + )?)), + None => Ok(None), + } + } + + #[cfg(not(any(test, feature = "testing")))] + fn acme_directory_ca(_cli: &Cli) -> anyhow::Result>> { + Ok(None) + } + + // One certificate source, or none. There is no third: a certificate the + // runtime minted for itself would be one an operator could mint too. + let acme = match cli.tls { + TlsMode::Off => None, + TlsMode::Acme => Some(enclave_runtime::serve::acme::start( + &AcmeConfig { + domains: cli.tls_domains.clone(), + contacts: cli.acme_contacts.clone(), + directory: cli.acme_directory.clone(), + directory_ca: acme_directory_ca(&cli)?, + prefix: mounted.bucket_prefix.clone(), + }, + mounted.data.clone(), + mounted.keys.clone(), + )?), + }; + tracing::info!( + variables = env.len(), + listen = %cli.http_listen, + tls = ?cli.tls, + "serving guest" + ); + + // Not a choice. An enclave nobody can verify is an enclave for nothing, and + // a flag that turns verification off is a flag that can be turned off by + // whoever starts the process — which inside an enclave is the party the + // enclave exists to exclude. + let attestation = Some(entropy.clone()); + + // Per-client views of the one filesystem. Each client's guest sees + // its own directory as `/`; the mount, the block cache and the + // transaction stream are shared, so a client costs a directory + // rather than a mount. + // + // A client is serialised against itself and nobody else, so two + // clients can execute guest code at the same time. A guest is + // written for that; it is not a deployment decision. + let tenancy = { + let limits = enclave_runtime::PoolLimits { + max_tenants: cli.max_tenants, + idle_timeout: Duration::from_secs(cli.tenant_idle_secs), + max_requests_per_instance: cli.max_requests_per_instance, + }; + tracing::info!( + max_tenants = limits.max_tenants, + "serving each client its own directory of the filesystem" + ); + Some(std::sync::Arc::new(enclave_runtime::Tenancy::new(limits))) + }; + + // WebAuthn, when the deployment named a relying party. Both the + // id and the origin or neither: a gate with no origin to compare + // against would accept assertions from any page that could reach + // it. + // + // Empty entries are dropped, not refused: an image built with no + // extra origins still sets the variable, to nothing. + let allowed_origins: Vec = cli + .webauthn_allowed_origins + .iter() + .map(|o| o.trim()) + .filter(|o| !o.is_empty()) + .map(str::to_string) + .collect(); + // Parsed before anything serves: a malformed origin refuses the boot rather + // than quietly admitting less, or more, than the image says. + let egress = enclave_runtime::serve::EgressPolicy::from_allowlist( + enclave_runtime::serve::EgressAllowlist::parse(&cli.guest_egress_origins) + .context("parsing --guest-egress-origin")?, + ); + if let enclave_runtime::serve::EgressPolicy::Allowlist(list) = &egress { + let origins: Vec = list.origins().map(|o| o.to_string()).collect(); + tracing::warn!(?origins, "guests may send requests to these origins"); + } + + let authentication = match (&cli.webauthn_rp_id, &cli.webauthn_origin) { + (Some(rp_id), Some(origin)) => { + let credentials = + std::sync::Arc::new(enclave_runtime::FilesystemCredentials::new(fs.clone())); + let gate = std::sync::Arc::new(enclave_runtime::Gate::new( + enclave_runtime::build_relying_party(rp_id, origin, &allowed_origins)?, + enclave_runtime::ChallengeStore::new( + Duration::from_secs(cli.challenge_ttl_secs), + enclave_runtime::DEFAULT_CAPACITY, + ), + credentials.clone(), + enclave_runtime::TokenStore::new( + Duration::from_secs(cli.interaction_token_ttl_secs), + enclave_runtime::DEFAULT_TOKEN_CAPACITY, + ), + )); + let auth = std::sync::Arc::new(enclave_runtime::AuthEndpoints::new( + gate.clone(), + credentials, + fs.clone(), + entropy.clone(), + )); + tracing::info!( + rp_id, + origin, + allowed_origins = ?allowed_origins, + "registration is open to anyone; requiring a passkey assertion per request" + ); + Some((auth, gate)) + } + (None, None) => { + anyhow::ensure!( + allowed_origins.is_empty(), + "--webauthn-allowed-origin was given without --webauthn-rp-id and \ + --webauthn-origin. With no relying party, authentication is off and \ + the origin would be silently ignored." + ); + None + } + _ => anyhow::bail!( + "--webauthn-rp-id and --webauthn-origin must be given together. A gate \ + with no origin to compare against would accept an assertion from any \ + page that could reach it." + ), + }; + + // The console always, and CloudWatch alongside it when configured. + // Additive on purpose: the console is often the only thing working + // while a deployment is being brought up, and the destination that + // needs the network is exactly the part that may not be. + // + // TracingLogSink first, so console output is never queued behind + // the network sink. + let mut sinks: Vec> = + vec![std::sync::Arc::new(enclave_runtime::TracingLogSink)]; + let mut log_forwarder = None; + + if let Some(config) = cli.guest_log_config()? { + // Which image is writing to this stream. A fixed stream name is + // shared by every boot and every image, so without this nothing + // in it says which enclave produced which line. + let image = attestation + .as_ref() + .and_then(|nsm| match nsm.describe_pcr(0) { + Ok(pcr) => Some(hex::encode(pcr.value)), + Err(e) => { + tracing::warn!(error = %e, "could not read PCR0 for the guest log stream"); + None + } + }); + + // Bounded, like everything else on this path. Building the + // client resolves a credential provider, and that reaches IMDS + // through gvproxy — a path this deployment has not verified. A + // logging client must never be what decides whether the enclave + // binds its listener. + let destination: Option> = + match tokio::time::timeout( + enclave_runtime::STARTUP_PROBE_TIMEOUT, + enclave_runtime::CloudWatchDestination::connect(&config), + ) + .await + { + Ok(destination) => Some(std::sync::Arc::new(destination)), + Err(_) => { + tracing::error!( + timeout = ?enclave_runtime::STARTUP_PROBE_TIMEOUT, + log_group = %config.log_group, + "could not build a CloudWatch client in time; guest output \ + stays on the console only. Check that the enclave can reach \ + IMDS through gvproxy." + ); + None + } + }; + + // The boot marker is the probe. A definitive refusal — the + // stream is not there, or this identity may not write to it — + // stops the boot: it is a deployment mistake, and discovering + // it only from a console line that was meant to be shipped off + // the box would mean running blind indefinitely. Anything + // transient does not stop the boot, because CloudWatch having a + // bad day must not be a cosigner outage. + if let Some(destination) = destination { + // Carried to the forwarder when the probe did not manage to + // write it, so a stream that recovers still ends up saying + // which image is writing to it. `None` once it is written. + let mut marker = None; + match enclave_runtime::open_guest_log_stream( + &destination, + image.as_deref(), + &config.region, + ) + .await + { + Ok(()) => tracing::info!( + log_group = %config.log_group, + log_stream = %config.log_stream, + image = image.as_deref().unwrap_or("(unknown)"), + "guest output is going to CloudWatch" + ), + Err(enclave_runtime::PutError::Definitive(e)) => anyhow::bail!( + "guest logging is configured for {}/{} but that destination refused \ + us: {e}. The log group and stream must exist and this enclave's \ + credentials must carry logs:PutLogEvents for them. Leave \ + --guest-log-group unset for console-only logging.", + config.log_group, + config.log_stream, + ), + Err(enclave_runtime::PutError::Transient(e)) => { + tracing::warn!( + error = %e, + log_group = %config.log_group, + "could not reach CloudWatch at startup; guest logs will retry" + ); + marker = Some(enclave_runtime::guest_log_boot_marker( + image.as_deref(), + &config.region, + )); + } + } + + let (sink, forwarder) = enclave_runtime::start_guest_log_forwarder(destination, marker); + sinks.push(std::sync::Arc::new(sink)); + log_forwarder = Some(forwarder); + } + } + + // Resolved before the listener binds, so a credential that cannot be + // fetched is a boot failure rather than a surprise on the first wake. + let notify = match cli.notify_settings()? { + None => None, + Some(settings) => { + let service_account = match settings.credential { + FcmCredential::Account(account) => *account, + FcmCredential::Parameter(name) => { + let json = read_ssm_parameter(&cli, &name).await?; + enclave_runtime::ServiceAccount::parse(&json).with_context(|| { + format!("the FCM service account in SSM parameter {name}") + })? + } + }; + tracing::info!( + project = %settings.project_id, + "notifications are enabled; wake signals carry no content" + ); + Some(enclave_runtime::NotifyConfig { + project_id: settings.project_id, + service_account, + endpoint: settings.endpoint, + }) + } + }; + + // Before the listener binds, so no request can produce guest output + // before there is a task consuming it. Held for the server's + // lifetime and drained explicitly below. + let (guest_logs, log_collector) = enclave_runtime::guest_io::start(std::sync::Arc::new( + enclave_runtime::FanOutSink::new(sinks), + )); + + let guest = GuestEnvironment::new(fs, clock, entropy, &env, &cli.guest_args, guest_logs)?; + let served = serve_component( + &component, + guest, + ServeConfig { + notify, + background_tasks: cli + .background_tasks + .then(|| enclave_runtime::tasks::TaskLimits { + concurrency: cli.background_concurrency, + timeout: Duration::from_secs(cli.background_timeout_secs), + max_records: cli.background_max_records, + per_tenant: cli.background_per_tenant, + }), + addr: cli.http_listen, + certificate: None, + acme, + attestation, + request_timeout: Duration::from_secs(cli.request_timeout_secs), + max_interaction: Duration::from_secs(cli.max_interaction_secs), + tenancy, + authentication, + egress, + }, + ) + .await; + // Drained whether the accept loop stopped cleanly or failed, and + // on a bounded deadline — a log that will not flush must not be + // what keeps an enclave from stopping. + log_collector.shutdown().await; + // After the collector, so records it was still holding reach the + // forwarder before the forwarder is asked to flush. Both deadlines + // are bounded; neither can hold the enclave open. + if let Some(forwarder) = log_forwarder { + forwarder.shutdown().await; + } + served?; + // `serve_component` only returns on error; reaching here means the + // accept loop stopped, which is not a guest exit. + Ok(enclave_runtime::GuestOutcome::Failed) +} + +/// Print what the configured clock actually reports. +/// +/// The delta against `CLOCK_REALTIME` is the interesting column: a PTP clock +/// disciplined by an external source and a host clock set by the hypervisor +/// have no reason to agree, and how far apart they are is exactly what this +/// feature exists to expose. +fn clock_check(clock: &dyn enclave_runtime::TrustedClock) -> Result<()> { + use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; + + println!("clock source: {}", clock.describe()); + println!("resolution: {:?}", clock.resolution()); + println!(); + println!( + " {:<26} {:>16} {:>12}", + "reading", "vs CLOCK_REALTIME", "read time" + ); + + let mut previous: Option = None; + for _ in 0..5 { + let started = Instant::now(); + let now = clock.now()?; + let read_time = started.elapsed(); + let realtime = SystemTime::now() + .duration_since(UNIX_EPOCH) + .expect("host clock before the epoch"); + + // Signed skew, in milliseconds. + let skew_ms = now.as_secs_f64() * 1e3 - realtime.as_secs_f64() * 1e3; + + if let Some(prev) = previous { + if now < prev { + anyhow::bail!("clock went backwards: {:?} then {:?}", prev, now); + } + } + previous = Some(now); + + println!( + " {:<26} {:>13.3} ms {:>9.1} us", + format!("{}.{:09}", now.as_secs(), now.subsec_nanos()), + skew_ms, + read_time.as_secs_f64() * 1e6 + ); + std::thread::sleep(Duration::from_millis(20)); + } + + println!(); + println!("clock advanced monotonically across 5 readings"); + Ok(()) +} + +fn exit_code(code: i32) -> std::process::ExitCode { + // `ExitCode` is a byte; a guest exit code outside that range would wrap + // silently, so clamp it to something a parent can read unambiguously. + std::process::ExitCode::from(u8::try_from(code).unwrap_or(1)) +} + +fn init_tracing() { + tracing_subscriber::fmt() + .with_env_filter( + tracing_subscriber::EnvFilter::try_from_default_env() + .unwrap_or_else(|_| tracing_subscriber::EnvFilter::new("info")), + ) + // Targets are shown, and that is not cosmetic. The console carries the + // runtime's own events *and* whatever a guest chose to write, and the + // target is the one thing separating them that a guest cannot + // influence — see `enclave_runtime::guest_io`. Guest lines say + // `guest`; everything else names a module in this runtime. Hiding it + // was right when every line was the runtime's own; it stopped being + // right when untrusted text started sharing the same console. + // + // `RUST_LOG=guest=warn` filters guest output independently either way, + // but an operator reading the console should not have to know that to + // tell which lines are trustworthy. + .with_target(true) + .compact() + .init(); +} + +/// Report on the entropy source, and apply the crude checks that catch a +/// device returning constants. +/// +/// These prove nothing about randomness quality — no cheap test can — but they +/// do catch the failure modes that actually occur: a stub that returns zeros, a +/// buffer never written, a device answering the same block every time. That is +/// exactly how an emulator or a misconfigured driver misbehaves. +fn entropy_check(source: &dyn enclave_runtime::Nsm) -> Result<()> { + use std::time::Instant; + + println!("entropy source: {}", source.describe()); + + let mut first = [0u8; 64]; + let started = Instant::now(); + source.get_random(&mut first)?; + let first_read = started.elapsed(); + + let mut second = [0u8; 64]; + source.get_random(&mut second)?; + + if first.iter().all(|&b| b == 0) { + anyhow::bail!("entropy source returned all zeros"); + } + if first.iter().all(|&b| b == first[0]) { + anyhow::bail!("entropy source returned a constant byte {:#04x}", first[0]); + } + if first == second { + anyhow::bail!("two draws returned identical bytes; the source is not advancing"); + } + + // A byte histogram over a larger sample. Not a randomness test — with 4096 + // samples over 256 buckets the mean is 16, and anything above ~80 in one + // bucket is a stuck source rather than bad luck. + let mut bulk = vec![0u8; 4096]; + let bulk_started = Instant::now(); + source.get_random(&mut bulk)?; + let bulk_read = bulk_started.elapsed(); + + let mut histogram = [0u32; 256]; + for &b in &bulk { + histogram[b as usize] += 1; + } + let peak = histogram.iter().copied().max().unwrap_or(0); + let empty = histogram.iter().filter(|&&c| c == 0).count(); + if peak > 80 { + anyhow::bail!("byte histogram peaks at {peak} of 4096; the source looks stuck"); + } + + println!( + " sample: {}", + first[..16] + .iter() + .map(|b| format!("{b:02x}")) + .collect::>() + .join("") + ); + println!(" histogram: peak {peak}, {empty} of 256 values unseen (mean 16)"); + println!( + " read cost: {:.1} us for 64 bytes, {:.1} us for 4096", + first_read.as_secs_f64() * 1e6, + bulk_read.as_secs_f64() * 1e6 + ); + println!(); + println!("entropy source produced varying, non-constant bytes"); + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use clap::CommandFactory; + + /// The minimum a parse needs, so a test can say only what it is about. + /// + /// `--master-key-source` is here because it has no default and never + /// should: an enclave's key handling is not something to inherit by + /// omission. Tests that care which source is chosen pass their own. + fn cli_from(args: &[&str]) -> Cli { + let mut full = vec!["enclave-runtime", "--master-key-source", "static"]; + full.extend_from_slice(args); + Cli::try_parse_from(full).expect("parse") + } + + /// The S3 timeout is its own setting, and reaches the mount config. It + /// used to be hardcoded here *and* ignored by the backend, so neither half + /// of the path worked. + #[test] + fn the_s3_timeout_is_configurable_and_separate_from_the_guest_one() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--s3-timeout-secs", + "9", + "--request-timeout-secs", + "45", + ]); + assert_eq!( + cli.mount_config().expect("valid").request_timeout, + Duration::from_secs(9) + ); + // Changing the guest deadline must not move the S3 one. + assert_eq!(cli.request_timeout_secs, 45); + } + + /// Console-only is the default, and takes no client and no credentials. + #[test] + fn notifications_are_off_unless_configured() { + let cli = cli_from(&["--bucket", "b", "--master-key", &"aa".repeat(32)]); + assert!(cli.notify_settings().expect("valid").is_none()); + } + + fn account_json() -> String { + serde_json::json!({ + "type": "service_account", + "project_id": "enclave-test", + "private_key_id": "kid-1", + "private_key": include_str!("notify/testdata/service-account-key.pem"), + "client_email": "wake@enclave-test.iam.gserviceaccount.com", + }) + .to_string() + } + + /// An image carries the setting; a deployment blanks it. Empty is off, not + /// malformed — the same rule the guest log settings follow. + #[test] + fn an_empty_setting_turns_notifications_off() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--fcm-project-id", + "", + "--fcm-service-account", + "", + ]); + assert!(cli.notify_settings().expect("valid").is_none()); + } + + #[test] + fn a_literal_service_account_is_parsed_at_startup_rather_than_on_first_use() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--fcm-project-id", + "enclave-test", + "--fcm-service-account", + &account_json(), + ]); + let settings = cli.notify_settings().expect("valid").expect("configured"); + assert_eq!(settings.project_id, "enclave-test"); + assert!(matches!(settings.credential, FcmCredential::Account(_))); + + // And a broken one fails here, not later. + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--fcm-project-id", + "enclave-test", + "--fcm-service-account", + "{\"type\":\"service_account\"}", + ]); + assert!(cli.notify_settings().is_err()); + } + + /// Refused rather than resolved by precedence: a deployment that set both + /// has one of them wrong, and guessing which is the wrong kind of help. + #[test] + fn two_credential_sources_are_refused_rather_than_ranked() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--fcm-project-id", + "p", + "--fcm-service-account", + &account_json(), + "--fcm-service-account-parameter", + "/prod/fcm", + ]); + assert!(cli.notify_settings().is_err()); + } + + #[test] + fn a_half_configured_notifier_is_refused_without_touching_the_network() { + let project_only = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--fcm-project-id", + "p", + ]); + assert!(project_only.notify_settings().is_err()); + + let credential_only = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--fcm-service-account-parameter", + "/prod/fcm", + ]); + assert!(credential_only.notify_settings().is_err()); + } + + #[test] + fn guest_logging_is_console_only_unless_configured() { + let cli = cli_from(&["--bucket", "b", "--master-key", &"aa".repeat(32)]); + assert!(cli.guest_log_config().expect("valid").is_none()); + } + + /// An empty setting means off, not malformed. + /// + /// The image environment is layered: the QEMU image inherits production's + /// and sets what it cannot use to "". Treating that as a typo made the + /// harness enclave refuse to boot, which is how this rule was learned. + #[test] + fn an_empty_setting_turns_guest_logging_off() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--guest-log-group", + "", + "--guest-log-stream", + "", + ]); + assert!( + cli.guest_log_config() + .expect("empty is off, not an error") + .is_none(), + "an empty group should mean console-only" + ); + } + + /// Half a destination is a typo, and knowable without touching the network. + #[test] + fn a_log_group_without_a_stream_is_refused() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--guest-log-group", + "/enclave/guest", + ]); + let error = cli.guest_log_config().unwrap_err(); + let message = format!("{error:#}"); + assert!(message.contains("--guest-log-stream"), "{message}"); + + // And the other way round. + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--guest-log-stream", + "guest", + ]); + assert!(cli.guest_log_config().is_err()); + } + + #[test] + fn a_configured_destination_carries_the_settings_through() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--region", + "eu-west-2", + "--guest-log-group", + "/enclave/guest", + "--guest-log-stream", + "guest", + "--guest-log-endpoint", + "http://127.0.0.1:4566", + ]); + let config = cli.guest_log_config().expect("valid").expect("configured"); + assert_eq!(config.log_group, "/enclave/guest"); + assert_eq!(config.log_stream, "guest"); + assert_eq!(config.region, "eu-west-2"); + assert_eq!(config.endpoint.as_deref(), Some("http://127.0.0.1:4566")); + } + + #[test] + fn the_cli_definition_is_valid() { + Cli::command().debug_assert(); + } + + /// There is no default guest. The image used to carry one at a fixed path; + /// it carries a key in the roots bucket instead, and a runtime given + /// neither says what is missing rather than guessing. + #[test] + fn a_guest_source_is_required() { + let cli = cli_from(&["--bucket", "b", "--master-key", &"aa".repeat(32)]); + let err = cli.guest_source().unwrap_err(); + assert!(format!("{err:#}").contains("--guest-object"), "{err:#}"); + } + + #[test] + fn a_guest_object_is_a_key_in_the_roots_bucket() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--guest-object", + "guest/guest.wasm", + ]); + assert_eq!( + cli.guest_source().unwrap(), + GuestSource::Object { + key: "guest/guest.wasm".into() + } + ); + } + + #[test] + fn a_guest_path_is_for_local_runs() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--guest-path", + "/tmp/other.wasm", + ]); + assert_eq!( + cli.guest_source().unwrap(), + GuestSource::Path(PathBuf::from("/tmp/other.wasm")) + ); + } + + #[test] + fn two_guest_sources_are_refused() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--guest-object", + "guest/guest.wasm", + "--guest-path", + "/tmp/other.wasm", + ]); + assert!(cli.guest_source().is_err()); + } + + /// The image environment is layered, and "" is how a layer says "not this + /// one" — the same rule as the guest log settings. + #[test] + fn an_empty_guest_setting_counts_as_unset() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--guest-object", + "", + "--guest-path", + "/tmp/other.wasm", + ]); + assert_eq!( + cli.guest_source().unwrap(), + GuestSource::Path(PathBuf::from("/tmp/other.wasm")) + ); + } + + /// The explicit model, which is what `--no-inherit-env` leaves you with: + /// the guest gets exactly what is named and nothing else. This was + /// `s3fs-runner`'s whole reason to exist before it was deleted, so it is + /// tested here rather than assumed. + #[test] + fn named_variables_reach_the_guest_when_nothing_is_inherited() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--no-inherit-env", + "--guest-env", + "LOG_LEVEL=debug", + ]); + assert_eq!( + cli.env_policy().build().unwrap(), + vec![("LOG_LEVEL".to_string(), "debug".to_string())] + ); + } + + /// Repeatable *and* comma-separated, because inside an enclave the setting + /// arrives as one `S3FS_GUEST_ENV` string and there is nowhere to repeat a + /// flag from. + #[test] + fn guest_env_accepts_a_comma_separated_list() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--no-inherit-env", + "--guest-env", + "A=1,B=2", + ]); + let env = cli.env_policy().build().unwrap(); + assert_eq!( + env, + vec![ + ("A".to_string(), "1".to_string()), + ("B".to_string(), "2".to_string()) + ] + ); + } + + #[test] + fn the_bucket_and_key_source_are_required() { + assert!(Cli::try_parse_from(["enclave-runtime"]).is_err()); + assert!(Cli::try_parse_from(["enclave-runtime", "--bucket", "b"]).is_err()); + // A bucket without a key source is not enough: there is no default, + // and picking one would mean guessing at how the enclave gets its key. + assert!(Cli::try_parse_from([ + "enclave-runtime", + "--bucket", + "b", + "--master-key", + &"ab".repeat(32), + ]) + .is_err()); + } + + fn key_config(args: &[&str]) -> Result { + let mut full = vec!["--bucket", "b"]; + full.extend_from_slice(args); + Cli::try_parse_from(std::iter::once("enclave-runtime").chain(full.iter().copied())) + .expect("parse") + .master_key_config() + } + + fn source_error(args: &[&str]) -> String { + let config = key_config(args).expect("config"); + let nsm = std::sync::Arc::new(nitro_nsm::fake::FakeNsm::new()); + format!( + "{:#}", + enclave_runtime::open_key_source(&config, nsm) + .expect_err("this combination must be refused") + ) + } + + /// The refusal that matters most. A production image that still carried + /// `S3FS_MASTER_KEY` would work perfectly — quietly taking its key from + /// the parent instance, which is the whole thing KMS release prevents. + #[test] + fn a_plaintext_key_is_refused_under_kms() { + let err = source_error(&[ + "--master-key-source", + "kms", + "--kms-key-id", + "arn:aws:kms:eu-west-2:1:key/a", + "--master-key-parameter", + "/p", + "--master-key", + &"ab".repeat(32), + ]); + assert!(err.contains("refused"), "unexpected error: {err}"); + } + + /// The mirror image: KMS settings under `static` mean one of the two is + /// not what was meant, and there is no safe way to guess which. + #[test] + fn kms_settings_are_refused_under_static() { + let err = source_error(&[ + "--master-key-source", + "static", + "--master-key", + &"ab".repeat(32), + "--kms-key-id", + "arn:aws:kms:eu-west-2:1:key/a", + ]); + assert!(err.contains("alongside KMS settings"), "unexpected: {err}"); + } + + #[test] + fn each_source_names_the_setting_it_is_missing() { + assert!(source_error(&["--master-key-source", "static"]).contains("--master-key")); + assert!(source_error(&["--master-key-source", "kms"]).contains("--kms-key-id")); + assert!(source_error(&[ + "--master-key-source", + "kms", + "--kms-key-id", + "arn:aws:kms:eu-west-2:1:key/a", + ]) + .contains("--master-key-parameter")); + } + + /// Warm clients are bounded by default. Each costs a wasm linear memory, + /// so an unbounded pool is an enclave that dies of memory exhaustion under + /// the load it was built for. + #[test] + fn the_warm_client_count_is_bounded_by_default() { + let cli = cli_from(&["--bucket", "b"]); + assert_eq!(cli.max_tenants, 64); + assert!(cli.tenant_idle_secs > 0, "idle tenants are never reclaimed"); + } + + /// The encryption context is built from the filesystem id and the + /// environment, so it is the same at mint and at open by construction. + #[test] + fn the_key_config_carries_the_encryption_context_inputs() { + let config = key_config(&[ + "--master-key-source", + "kms", + "--fs-id", + &"cd".repeat(16), + "--environment", + "staging", + ]) + .expect("config"); + assert_eq!(config.environment, "staging"); + assert_eq!(config.fs_id, [0xcd; 16]); + } + + #[test] + fn guest_arguments_come_after_a_separator() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--", + "arg1", + "--not-our-flag", + ]); + assert_eq!(cli.guest_args, vec!["arg1", "--not-our-flag"]); + } + + /// The default is inheritance, because a guest that needs configuration + /// should get it without every variable being enumerated. + #[test] + fn the_environment_is_inherited_by_default() { + let cli = cli_from(&["--bucket", "b", "--master-key", &"aa".repeat(32)]); + assert!(cli.env_policy().inherit); + } + + #[test] + fn no_inherit_env_switches_to_allowlist_only() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--no-inherit-env", + "--guest-env", + "RUST_LOG=debug", + ]); + let policy = cli.env_policy(); + assert!(!policy.inherit); + assert_eq!(policy.explicit, vec!["RUST_LOG=debug"]); + } + + #[test] + fn an_invalid_filesystem_id_is_rejected_before_anything_is_opened() { + let cli = cli_from(&[ + "--bucket", + "b", + "--master-key", + &"aa".repeat(32), + "--fs-id", + "not-hex", + ]); + assert!(cli.mount_config().is_err()); + } + + #[test] + fn mount_config_carries_the_settings_through() { + let cli = cli_from(&[ + "--bucket", + "data", + "--roots-bucket", + "roots", + "--master-key", + &"aa".repeat(32), + "--bucket-prefix", + "tenant", + "--min-root-seq", + "42", + "--force-path-style", + ]); + let cfg = cli.mount_config().unwrap(); + assert_eq!(cfg.bucket, "data"); + assert_eq!(cfg.roots_bucket.as_deref(), Some("roots")); + assert_eq!(cfg.bucket_prefix, "tenant"); + assert_eq!(cfg.min_root_seq, Some(42)); + assert!(cfg.force_path_style); + } + + /// `ExitCode` is a byte. A guest exit code outside that range must not + /// wrap into something a parent would read as success. + #[test] + fn out_of_range_exit_codes_do_not_wrap_to_success() { + assert_eq!( + format!("{:?}", exit_code(256)), + format!("{:?}", exit_code(1)) + ); + assert_eq!( + format!("{:?}", exit_code(-1)), + format!("{:?}", exit_code(1)) + ); + assert_ne!( + format!("{:?}", exit_code(256)), + format!("{:?}", exit_code(0)) + ); + } +} diff --git a/runtime/src/mount.rs b/runtime/src/mount.rs new file mode 100644 index 0000000..dea05cc --- /dev/null +++ b/runtime/src/mount.rs @@ -0,0 +1,258 @@ +//! Turning configuration into a mounted filesystem. + +use std::sync::Arc; +use std::time::Duration; + +use anyhow::{Context, Result}; +use s3fs_core::backend::{AwsS3Backend, AwsS3BackendConfig, Backend}; +use s3fs_core::crypto::KeyMaterial; +use s3fs_core::{Config, Fs, MasterSecret}; + +/// Everything needed to open the store. +#[derive(Debug, Clone)] +pub struct MountConfig { + /// Bucket holding the data slabs. + pub bucket: String, + /// Bucket holding the signed root records. `None` uses `bucket`. + pub roots_bucket: Option, + pub region: String, + pub endpoint: Option, + pub access_key_id: Option, + pub secret_access_key: Option, + pub session_token: Option, + pub force_path_style: bool, + pub bucket_prefix: String, + pub mount_path: String, + /// Filesystem identifier, the key-derivation salt. + pub fs_id: [u8; 16], + /// Refuse to mount a root older than this. + pub min_root_seq: Option, + /// Skip the `HeadBucket` startup probe. + pub skip_bucket_probe: bool, + pub request_timeout: Duration, +} + +impl MountConfig { + fn backend_config(&self, bucket: &str) -> AwsS3BackendConfig { + AwsS3BackendConfig { + bucket: bucket.to_string(), + region: self.region.clone(), + endpoint: self.endpoint.clone(), + access_key_id: self.access_key_id.clone(), + secret_access_key: self.secret_access_key.clone(), + session_token: self.session_token.clone(), + force_path_style: self.force_path_style, + request_timeout: self.request_timeout, + } + } +} + +/// Parse the 32-hex-character filesystem identifier. +pub fn parse_fs_id(s: &str) -> Result<[u8; 16]> { + let s = s.trim(); + if s.len() != 32 { + anyhow::bail!("filesystem id must be 32 hex characters, got {}", s.len()); + } + let mut out = [0u8; 16]; + for (i, byte) in out.iter_mut().enumerate() { + *byte = u8::from_str_radix(&s[i * 2..i * 2 + 2], 16) + .map_err(|_| anyhow::anyhow!("filesystem id is not valid hex"))?; + } + Ok(out) +} + +/// A mounted filesystem, plus the pieces a runtime needs alongside it. +/// +/// `Debug` names the store rather than dumping the filesystem: an `Fs` renders +/// its whole handle table, which is noise in a boot log and unbounded in a +/// panic message. +/// +/// The ACME cache writes sealed objects to the data bucket and seals them +/// under a key derived from the same master secret, so it needs both — but it +/// deliberately does *not* go through the filesystem, because the guest's +/// preopen is the filesystem root and the blob contains a TLS private key. +pub struct Mounted { + pub fs: Arc, + pub data: Arc, + pub keys: Arc, + pub bucket_prefix: String, +} + +impl std::fmt::Debug for Mounted { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("Mounted") + .field("bucket_prefix", &self.bucket_prefix) + .finish_non_exhaustive() + } +} + +/// Both backends, connected but not yet mounted. +/// +/// Separate from mounting because the boot machine has to *read* the store — +/// the state-origin receipt lives in the roots bucket — before it can decide +/// whether this is a genesis or a resume, and only after deciding does it know +/// which master secret to use. +pub struct Backends { + pub data: Arc, + pub roots: Arc, +} + +/// Connect to both buckets. +pub async fn connect(config: &MountConfig) -> Result { + let data_cfg = config.backend_config(&config.bucket); + let data: Arc = Arc::new(if config.skip_bucket_probe { + AwsS3Backend::connect_unchecked(data_cfg).await? + } else { + AwsS3Backend::connect(data_cfg) + .await + .context("connecting to the data bucket")? + }); + + let roots_name = config.roots_bucket.as_deref().unwrap_or(&config.bucket); + let roots: Arc = if roots_name == config.bucket { + // One bucket for both. Simpler to operate, but the anchor and the + // reclaimable data then share a retention policy, which defeats the + // point of splitting them. + tracing::warn!( + bucket = %config.bucket, + "roots and data share a bucket; Object Lock retention cannot then \ + differ between the rollback anchor and reclaimable block storage" + ); + data.clone() + } else { + Arc::new(AwsS3Backend::connect_unchecked(config.backend_config(roots_name)).await?) + }; + + Ok(Backends { data, roots }) +} + +/// Mount an existing filesystem with a secret the caller has already resolved. +pub async fn mount_existing( + backends: &Backends, + config: &MountConfig, + master: &MasterSecret, +) -> Result { + finish(backends, config, master, false).await +} + +/// Create a filesystem. Only genesis calls this. +pub async fn create( + backends: &Backends, + config: &MountConfig, + master: &MasterSecret, +) -> Result { + finish(backends, config, master, true).await +} + +async fn finish( + backends: &Backends, + config: &MountConfig, + master: &MasterSecret, + genesis: bool, +) -> Result { + let data = backends.data.clone(); + let roots = backends.roots.clone(); + + let fs_config = Config::builder() + .bucket_prefix(config.bucket_prefix.clone()) + .mount_path(config.mount_path.clone()) + .build(); + + // Derived twice — once here and once inside `Fs::mount` — rather than + // reaching into the store for them. The derivation is cheap and pure, and + // threading them out would widen the store's API for one caller. + let derived = Arc::new( + KeyMaterial::derive(master, config.fs_id) + .map_err(|e| anyhow::anyhow!("deriving keys: {e}"))?, + ); + + let fs = if genesis { + Fs::create( + data.clone(), + roots, + master, + config.fs_id, + Arc::new(fs_config), + ) + .await + .map_err(|e| anyhow::anyhow!("creating filesystem: {e}"))? + } else { + Fs::mount( + data.clone(), + roots, + master, + config.fs_id, + Arc::new(fs_config), + config.min_root_seq, + ) + .await + .map_err(|e| anyhow::anyhow!("mounting filesystem: {e}"))? + }; + + // The operator's evidence of which committed state this process came up + // on. Under attestation this is what the health endpoint reports, and it + // is the number to compare against an external freshness floor. + let root = fs + .store() + .root() + .await + .map_err(|e| anyhow::anyhow!("reading root record: {e}"))?; + tracing::info!( + mode = if genesis { "genesis" } else { "resume" }, + root_seq = root.seq, + txg = root.txg, + merkle_root = %root.merkle_root(), + data_bucket = %config.bucket, + roots_bucket = %config.roots_bucket.as_deref().unwrap_or(&config.bucket), + "mounted" + ); + + Ok(Mounted { + fs, + data, + keys: derived, + bucket_prefix: config.bucket_prefix.clone(), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn fs_id_round_trips() { + let id = parse_fs_id("000102030405060708090a0b0c0d0e0f").unwrap(); + assert_eq!(id[0], 0x00); + assert_eq!(id[15], 0x0f); + assert!(parse_fs_id(" 00000000000000000000000000000000\n").is_ok()); + } + + #[test] + fn fs_id_rejects_bad_input() { + assert!(parse_fs_id("abcd").is_err()); + assert!(parse_fs_id(&"z".repeat(32)).is_err()); + assert!(parse_fs_id("").is_err()); + } + + #[test] + fn the_roots_bucket_defaults_to_the_data_bucket() { + let cfg = MountConfig { + bucket: "data".into(), + roots_bucket: None, + region: "us-east-1".into(), + endpoint: None, + access_key_id: None, + secret_access_key: None, + session_token: None, + force_path_style: false, + bucket_prefix: String::new(), + mount_path: "/".into(), + fs_id: [0u8; 16], + min_root_seq: None, + skip_bucket_probe: true, + request_timeout: Duration::from_secs(30), + }; + assert_eq!(cfg.roots_bucket.as_deref().unwrap_or(&cfg.bucket), "data"); + assert_eq!(cfg.backend_config("roots").bucket, "roots"); + } +} diff --git a/runtime/src/net.rs b/runtime/src/net.rs new file mode 100644 index 0000000..9be9930 --- /dev/null +++ b/runtime/src/net.rs @@ -0,0 +1,259 @@ +//! Bringing up enclave networking over vsock. +//! +//! A Nitro enclave has no network interface. Its only channel to the outside +//! is `AF_VSOCK` to the parent instance, which means nothing that speaks TCP +//! — not the AWS SDK, not an ACME client, not a TLS listener — works until +//! something turns that channel into an interface. +//! +//! ```text +//! parent instance (untrusted) enclave (attested) +//! ┌──────────────────────────┐ ┌────────────────────────┐ +//! │ gvproxy │ vsock │ gvforwarder │ +//! │ --listen vsock://:1024 │◀───────▶│ -url vsock://3:1024 │ +//! │ 192.168.127.1 gw + DNS │ CID 3 │ tap0 192.168.127.2 │ +//! └──────────────────────────┘ :1024 └────────────────────────┘ +//! ``` +//! +//! `gvproxy` is a user-mode network stack: it terminates the enclave's +//! ethernet frames on the parent and forwards them, giving the enclave a +//! gateway, DHCP and DNS at `192.168.127.1`. This is what nitriding and +//! ArkLabs both do, and the reason is worth stating: with a real interface, +//! ordinary networking code works unmodified. The alternative — a vsock +//! connector threaded through the AWS SDK and the ACME client — means two +//! bespoke transports to write and maintain, and neither library expects it. +//! +//! ## What this does and does not give away +//! +//! The parent sees ciphertext. Traffic to S3, KMS and an ACME provider is TLS +//! from inside the enclave, terminated at the far end, so gvproxy carries +//! bytes it cannot read. What the parent *does* learn is metadata — who the +//! enclave talks to, when, and how much — and it can of course refuse to carry +//! anything. Neither is new: it already controls whether the enclave runs. +//! +//! What would be new, and is not done here, is trusting the parent's DNS. A +//! parent that answers DNS can point the enclave at its own endpoint, which is +//! precisely why every outbound connection is TLS with certificate validation +//! and why the S3 data is encrypted under keys the parent never holds. + +use std::net::{Ipv4Addr, SocketAddrV4, TcpStream}; +use std::path::PathBuf; +use std::process::{Child, Command, Stdio}; +use std::time::{Duration, Instant}; + +use anyhow::{Context, Result}; + +/// The parent instance's context ID. Fixed by AWS. +pub const PARENT_CID: u32 = 3; +/// Port `gvproxy` listens on for the enclave's frames, matching its own default. +pub const GVPROXY_PORT: u32 = 1024; +/// The gateway gvproxy presents. Also its DNS server and HTTP API. +pub const GATEWAY: Ipv4Addr = Ipv4Addr::new(192, 168, 127, 1); +/// The address gvproxy's DHCP hands the enclave. +pub const ENCLAVE_ADDRESS: Ipv4Addr = Ipv4Addr::new(192, 168, 127, 2); + +/// Where the forwarder binary lives inside the enclave image. +pub const DEFAULT_GVFORWARDER: &str = "/usr/local/bin/gvforwarder"; + +/// How the enclave reaches the network. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum NetworkMode { + /// Do nothing. Correct outside an enclave, where the host already has an + /// interface, and for a guest that needs no outbound access. + None, + /// Run `gvforwarder` against the parent's `gvproxy`. + Gvproxy, +} + +impl NetworkMode { + pub fn parse(s: &str) -> Result { + match s.trim().to_ascii_lowercase().as_str() { + "none" | "off" | "host" => Ok(NetworkMode::None), + "gvproxy" | "tap" | "vsock" => Ok(NetworkMode::Gvproxy), + other => Err(format!("expected one of none, gvproxy; got {other:?}")), + } + } +} + +/// A running `gvforwarder`, killed when dropped. +/// +/// Held rather than detached so that the forwarder dies with the runtime. A +/// surviving child would keep the tap device up around a runtime that had +/// exited, which on a restart looks like an interface that exists but carries +/// nothing. +#[derive(Debug)] +pub struct Network { + child: Option, +} + +impl Drop for Network { + fn drop(&mut self) { + if let Some(child) = &mut self.child { + let _ = child.kill(); + let _ = child.wait(); + } + } +} + +/// Configuration for [`bring_up`]. +#[derive(Debug, Clone)] +pub struct NetworkConfig { + pub mode: NetworkMode, + pub gvforwarder: PathBuf, + pub parent_cid: u32, + pub port: u32, + /// How long to wait for the gateway to answer before giving up. + pub timeout: Duration, +} + +impl Default for NetworkConfig { + fn default() -> Self { + NetworkConfig { + mode: NetworkMode::None, + gvforwarder: PathBuf::from(DEFAULT_GVFORWARDER), + parent_cid: PARENT_CID, + port: GVPROXY_PORT, + timeout: Duration::from_secs(30), + } + } +} + +/// Start the forwarder and wait until the network actually carries traffic. +/// +/// Returns `None` for [`NetworkMode::None`], so callers need no branch. +pub fn bring_up(config: &NetworkConfig) -> Result> { + if config.mode == NetworkMode::None { + return Ok(None); + } + + // DNS first: gvforwarder brings the interface up and takes an address by + // DHCP, but nothing writes resolv.conf, and a runtime that can route but + // not resolve fails later with errors that name the wrong problem. + write_resolv_conf().context("configuring DNS for the enclave")?; + + let url = format!("vsock://{}:{}/connect", config.parent_cid, config.port); + tracing::info!( + gvforwarder = %config.gvforwarder.display(), + %url, + "bringing up enclave networking" + ); + + let child = Command::new(&config.gvforwarder) + .arg("-url") + .arg(&url) + // No `-debug`: it dumps a decode of every frame, which buries the DHCP + // client's output — and the DHCP client is what usually fails. + // Inherited, not discarded. If the forwarder cannot create its tap + // device it says so and exits, and the only symptom visible from here + // is the gateway never answering — a timeout thirty seconds later + // that names the parent's gvproxy, which is usually running fine. + // Inside an enclave the console is the only place a diagnosis can go. + .stdout(Stdio::inherit()) + .stderr(Stdio::inherit()) + .spawn() + .with_context(|| { + format!( + "starting {} — it ships inside the enclave image, so a failure here \ + usually means the image was built without it", + config.gvforwarder.display() + ) + })?; + + let network = Network { child: Some(child) }; + + wait_for_gateway(config.timeout).with_context(|| { + format!( + "the enclave network never came up. The parent instance must be running \ + `gvproxy --listen vsock://:{}`; without it the forwarder connects to nothing.", + config.port + ) + })?; + + tracing::info!(address = %ENCLAVE_ADDRESS, gateway = %GATEWAY, "enclave networking is up"); + Ok(Some(network)) +} + +/// gvproxy answers DNS on the gateway address. +fn write_resolv_conf() -> Result<()> { + let contents = format!("nameserver {GATEWAY}\n"); + // An enclave image is whatever the ramdisk contains, and a minimal one may + // have no /etc at all — the failure is then `No such file or directory` + // against a path that looks like it must exist, several layers below + // anything that mentions DNS. + std::fs::create_dir_all("/etc").context("creating /etc")?; + std::fs::write("/etc/resolv.conf", contents).context("writing /etc/resolv.conf") +} + +/// Wait for the gateway to accept a connection. +/// +/// Readiness is a successful TCP connection to gvproxy's own HTTP API, not a +/// sleep: DHCP takes an unpredictable moment, and a fixed delay is either too +/// short on a slow boot or wasted on a fast one. gvproxy serves its API on the +/// gateway address, so an accepted connection means frames are crossing the +/// vsock in both directions. +fn wait_for_gateway(timeout: Duration) -> Result<()> { + let deadline = Instant::now() + timeout; + let target = SocketAddrV4::new(GATEWAY, 80); + let mut last: Option = None; + + while Instant::now() < deadline { + match TcpStream::connect_timeout(&target.into(), Duration::from_millis(500)) { + Ok(_) => return Ok(()), + Err(e) => last = Some(e), + } + std::thread::sleep(Duration::from_millis(200)); + } + + match last { + Some(e) => Err(anyhow::Error::from(e)) + .with_context(|| format!("gateway {GATEWAY} did not answer within {timeout:?}")), + None => anyhow::bail!("gateway {GATEWAY} did not answer within {timeout:?}"), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn modes_parse_the_documented_values() { + assert_eq!(NetworkMode::parse("none"), Ok(NetworkMode::None)); + assert_eq!(NetworkMode::parse("gvproxy"), Ok(NetworkMode::Gvproxy)); + assert_eq!(NetworkMode::parse("TAP"), Ok(NetworkMode::Gvproxy)); + assert!(NetworkMode::parse("bridge").is_err()); + } + + /// The default must do nothing. Tests and local runs happen outside an + /// enclave, where spawning a forwarder would fail — or, worse, succeed. + #[test] + fn the_default_touches_nothing() { + assert_eq!(NetworkConfig::default().mode, NetworkMode::None); + assert!(bring_up(&NetworkConfig::default()).unwrap().is_none()); + } + + /// The CID and port are fixed by AWS and by gvproxy respectively. Getting + /// either wrong produces a forwarder that connects to nothing and a + /// timeout thirty seconds later, so pin them. + #[test] + fn the_vsock_endpoint_is_the_one_gvproxy_listens_on() { + let config = NetworkConfig::default(); + assert_eq!(config.parent_cid, 3, "the parent instance is always CID 3"); + assert_eq!(config.port, 1024, "gvproxy's default vsock port"); + } + + #[test] + fn a_missing_forwarder_names_the_path_and_the_likely_cause() { + let config = NetworkConfig { + mode: NetworkMode::Gvproxy, + gvforwarder: PathBuf::from("/enclave/definitely-absent-forwarder"), + ..Default::default() + }; + // Writing /etc/resolv.conf may fail first when not running as root; + // either way the error must name something actionable. + let err = bring_up(&config).unwrap_err(); + let text = format!("{err:#}"); + assert!( + text.contains("definitely-absent-forwarder") || text.contains("resolv.conf"), + "unhelpful error: {text}" + ); + } +} diff --git a/runtime/src/notify/device.rs b/runtime/src/notify/device.rs new file mode 100644 index 0000000..7832fb4 --- /dev/null +++ b/runtime/src/notify/device.rs @@ -0,0 +1,768 @@ +//! Which devices a tenant can be woken on. +//! +//! A registration token is a capability: whoever holds one can wake that device +//! from anywhere, as this Firebase project. So tokens live at +//! `/runtime/devices//`, above every tenant scope and unreachable from +//! any guest — the same placement, and the same reason, as +//! [`crate::auth::credential`]. A guest enrols one and asks for a count; it +//! never reads one back. +//! +//! ## Why the filename is a hash +//! +//! A token's character set is Google's business and may widen. Naming the file +//! after the token would make path safety depend on a validator agreeing with +//! FCM forever; naming it `sha256(token)` makes the name safe by construction +//! and keeps the check that the record matches its own filename. +//! +//! ## Why the cap evicts instead of refusing +//! +//! At [`MAX_DEVICES_PER_TENANT`] a new enrolment drops the oldest rather than +//! failing. This is a cache of places a person can be reached, not a list of +//! credentials: somebody on their ninth phone must still be able to enrol it, +//! and the entry evicted is the one least likely to still be live. That is the +//! deliberate opposite of [`crate::auth::credential`], where a revoked passkey +//! is kept for ever so it can never be silently reinstated. + +use std::collections::BTreeMap; +use std::sync::Arc; + +use anyhow::{ensure, Context, Result}; +use s3fs_core::{Fs, FsError, Inode, InodeKind, OpenFlags}; +use serde::{Deserialize, Serialize}; +use tokio::sync::Mutex; + +use crate::auth::credential::RUNTIME_DIR; +use crate::tenant::ensure_dir; + +/// Where every tenant's devices live, one directory each. +pub const DEVICES_DIR: &str = "devices"; + +/// Devices one tenant may have enrolled at once. +pub const MAX_DEVICES_PER_TENANT: usize = 8; +/// Devices across every tenant, bounding what a recovery scan will load. +pub const MAX_DEVICES: usize = 4096; +/// An FCM registration token is ~160 characters today; these bound the shape +/// without pretending to know the format. +const MIN_TOKEN_LEN: usize = 32; +const MAX_TOKEN_LEN: usize = 512; +const MAX_RECORD: usize = 4 * 1024; +const RECORD_VERSION: u32 = 1; + +/// One enrolled device. +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] +pub struct StoredDevice { + /// An unknown version is a refusal, never a misparse. + pub version: u32, + /// Repeated inside the record so it can be checked against the directory + /// the record was found in. Neither check is sufficient alone: the filename + /// proves the token, the directory proves the owner. + pub tenant: [u8; 16], + pub token: String, + pub created_ms: u64, +} + +/// What the registry will hold. Injectable for the same reason `TaskLimits` is: +/// the caps are the interesting behaviour, and a test that had to write four +/// thousand records to reach one would not be written. +#[derive(Debug, Clone, Copy)] +pub struct DeviceLimits { + pub max_devices: usize, + pub per_tenant: usize, +} + +impl Default for DeviceLimits { + fn default() -> Self { + DeviceLimits { + max_devices: MAX_DEVICES, + per_tenant: MAX_DEVICES_PER_TENANT, + } + } +} + +/// Every tenant's devices, held in memory and backed by the filesystem. +/// +/// Held in memory because the forwarder needs a tenant's tokens on every wake, +/// and a `read_dir` per wake is a filesystem transaction — in production an S3 +/// round trip — in the path of a signal whose only value is being prompt. +pub struct DeviceRegistry { + fs: Arc, + dir: Arc, + limits: DeviceLimits, + devices: Mutex>>, +} + +impl std::fmt::Debug for DeviceRegistry { + /// Counts, never tokens. A token in a log line is a token in whatever reads + /// that log. + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("DeviceRegistry").finish_non_exhaustive() + } +} + +fn valid_token(token: &str) -> bool { + (MIN_TOKEN_LEN..=MAX_TOKEN_LEN).contains(&token.len()) + && token + .bytes() + .all(|b| b.is_ascii_alphanumeric() || matches!(b, b'-' | b'_' | b':' | b'.')) +} + +fn record_name(token: &str) -> String { + hex::encode(nitro_attestation::sha256(token.as_bytes())) +} + +fn is_hex(name: &str, len: usize) -> bool { + name.len() == len + && name + .bytes() + .all(|b| b.is_ascii_hexdigit() && !b.is_ascii_uppercase()) +} + +impl DeviceRegistry { + pub async fn open(fs: Arc) -> Result> { + Self::open_with_limits(fs, DeviceLimits::default()).await + } + + pub async fn open_with_limits(fs: Arc, limits: DeviceLimits) -> Result> { + anyhow::ensure!( + limits.max_devices > 0 && limits.per_tenant > 0, + "device limits must be positive" + ); + let runtime = ensure_dir(&fs, &fs.root(), RUNTIME_DIR).await?; + let dir = ensure_dir(&fs, &runtime, DEVICES_DIR).await?; + let registry = Arc::new(DeviceRegistry { + fs, + dir, + limits, + devices: Mutex::new(BTreeMap::new()), + }); + let loaded = registry.scan().await?; + *registry.devices.lock().await = loaded; + Ok(registry) + } + + /// Read every published record back, refusing anything that does not + /// account for itself. + async fn scan(&self) -> Result>> { + let mut devices: BTreeMap<[u8; 16], Vec> = BTreeMap::new(); + let mut total = 0usize; + + for tenant_entry in self.fs.read_dir(&self.dir).await? { + ensure!( + tenant_entry.kind == InodeKind::Directory && is_hex(&tenant_entry.name, 32), + "unexpected entry {:?} under /{RUNTIME_DIR}/{DEVICES_DIR}", + tenant_entry.name + ); + let mut tenant = [0u8; 16]; + hex::decode_to_slice(&tenant_entry.name, &mut tenant) + .context("decoding a tenant directory name")?; + let tenant_dir = self + .fs + .lookup_at(&self.dir, &tenant_entry.name) + .await + .map_err(|e| anyhow::anyhow!(e))?; + + for entry in self.fs.read_dir(&tenant_dir).await? { + // A write interrupted before its rename. Nothing published it, + // so nothing may read it. + if entry.name.ends_with(".tmp") { + self.fs.unlink(&tenant_dir, &entry.name).await?; + continue; + } + total += 1; + + let path = format!( + "/{RUNTIME_DIR}/{DEVICES_DIR}/{}/{}", + tenant_entry.name, entry.name + ); + let h = self.fs.open(&path, OpenFlags::read_only()).await?; + let bytes = self.fs.pread(&h, 0, MAX_RECORD + 1).await; + // Closed before the read is unwrapped, so a decode failure does + // not leak a handle on every attempt. + self.fs.close(&h).await?; + let bytes = bytes?; + ensure!(bytes.len() <= MAX_RECORD, "oversized device record"); + + let device: StoredDevice = + serde_json::from_slice(&bytes).context("decoding a device record")?; + ensure!( + device.version == RECORD_VERSION + && device.tenant == tenant + && valid_token(&device.token) + && record_name(&device.token) == entry.name, + "invalid device record at {path}" + ); + devices.entry(tenant).or_default().push(device); + } + } + + // Oldest first, so eviction and reporting have one order to rely on. + for list in devices.values_mut() { + list.sort_by_key(|d| (d.created_ms, d.token.clone())); + } + + // Over the cap, drop the oldest rather than refuse to start. + // + // A device record is a cache of where somebody can be reached, not work + // that must not be lost — so unlike a task record, the right answer to + // too many is to keep the newest and serve. Refusing would turn an + // over-full store into one that can never be booted to fix, and a store + // can be over-full legitimately: a lowered cap in a new image, or + // records written before the cap was enforced on the way in. + if total > self.limits.max_devices { + let mut all: Vec<([u8; 16], u64, String)> = devices + .iter() + .flat_map(|(tenant, list)| { + list.iter() + .map(move |d| (*tenant, d.created_ms, d.token.clone())) + }) + .collect(); + all.sort_by(|a, b| (a.1, &a.2).cmp(&(b.1, &b.2))); + let excess = total - self.limits.max_devices; + tracing::warn!( + total, + cap = self.limits.max_devices, + dropping = excess, + "the device store is over its cap; dropping the oldest records" + ); + for (tenant, _, token) in all.into_iter().take(excess) { + self.remove(tenant, &token).await?; + if let Some(list) = devices.get_mut(&tenant) { + list.retain(|d| d.token != token); + } + } + devices.retain(|_, list| !list.is_empty()); + } + + Ok(devices) + } + + /// Publish a record: temp file, committed, then renamed into place. + /// + /// The rename is what makes the record visible, so a crash leaves either + /// the old state or the new one and never a half-written record. + async fn write(&self, device: &StoredDevice) -> Result<()> { + let bytes = serde_json::to_vec(device)?; + ensure!(bytes.len() <= MAX_RECORD, "device record exceeds its limit"); + + let tenant_name = hex::encode(device.tenant); + let tenant_dir = ensure_dir(&self.fs, &self.dir, &tenant_name).await?; + let name = record_name(&device.token); + let temp = format!("{name}.tmp"); + + // Remove an unpublished write left by a cancelled host call. + match self.fs.unlink(&tenant_dir, &temp).await { + Ok(()) | Err(FsError::NotFound) => {} + Err(e) => return Err(e.into()), + } + let path = format!("/{RUNTIME_DIR}/{DEVICES_DIR}/{tenant_name}/{temp}"); + let h = self.fs.open(&path, OpenFlags::create_new()).await?; + let write = self.fs.pwrite(&h, 0, &bytes).await; + let close = self.fs.close(&h).await; + write?; + close?; // close commits the complete contents before publication + self.fs + .rename(&tenant_dir, &temp, &tenant_dir, &name) + .await?; + Ok(()) + } + + async fn remove(&self, tenant: [u8; 16], token: &str) -> Result<()> { + let tenant_name = hex::encode(tenant); + let Ok(tenant_dir) = self.fs.lookup_at(&self.dir, &tenant_name).await else { + return Ok(()); + }; + match self.fs.unlink(&tenant_dir, &record_name(token)).await { + Ok(()) | Err(FsError::NotFound) => Ok(()), + Err(e) => Err(anyhow::anyhow!(e)).context("removing a device record"), + } + } + + /// Enrol a token for this tenant. + /// + /// Enrolling a token the tenant already has is success and changes nothing, + /// including its `created_ms`: a client that re-registers on every launch — + /// which is what the FCM SDKs encourage — must not thereby keep itself at + /// the front of the eviction queue for ever. + pub async fn register(&self, tenant: [u8; 16], token: &str, now_ms: u64) -> Result<()> { + ensure!( + valid_token(token), + "a registration token must be {MIN_TOKEN_LEN}-{MAX_TOKEN_LEN} characters of \ + letters, digits, '-', '_', ':' or '.'" + ); + let mut devices = self.devices.lock().await; + if devices + .get(&tenant) + .is_some_and(|list| list.iter().any(|d| d.token == token)) + { + return Ok(()); + } + + // The global cap, enforced where the write happens — the same place + // `tasks::enqueue` enforces `max_records`. Without it, enrolments this + // call accepted could put the store past what a restart will load. + let total: usize = devices.values().map(Vec::len).sum(); + ensure!( + total < self.limits.max_devices, + "the device store is full ({} records); devices must be forgotten \ + before another can enrol", + self.limits.max_devices + ); + + let device = StoredDevice { + version: RECORD_VERSION, + tenant, + token: token.to_string(), + created_ms: now_ms, + }; + self.write(&device).await?; + + let list = devices.entry(tenant).or_default(); + list.push(device); + list.sort_by_key(|d| (d.created_ms, d.token.clone())); + // Published first, evicted second: a crash between the two leaves one + // device too many, which the next enrolment trims. The other order + // could leave the tenant with nothing enrolled at all. + while list.len() > self.limits.per_tenant { + let evicted = list.remove(0); + self.remove(tenant, &evicted.token).await?; + } + Ok(()) + } + + /// Forgetting a token this tenant does not have is success. + /// + /// A client that was pruned while it was offline is not wrong to ask. + pub async fn forget(&self, tenant: [u8; 16], token: &str) -> Result<()> { + let mut devices = self.devices.lock().await; + self.remove(tenant, token).await?; + if let Some(list) = devices.get_mut(&tenant) { + list.retain(|d| d.token != token); + } + Ok(()) + } + + /// Remove a token only if it is still the record we failed to reach. + /// + /// The forwarder decides to prune and applies it later, and in between an + /// interactive call may have re-enrolled the same token. Comparing + /// `created_ms` makes that a no-op rather than deleting a live enrolment. + pub async fn forget_if_unchanged( + &self, + tenant: [u8; 16], + token: &str, + created_ms: u64, + ) -> Result { + let mut devices = self.devices.lock().await; + let Some(list) = devices.get_mut(&tenant) else { + return Ok(false); + }; + if !list + .iter() + .any(|d| d.token == token && d.created_ms == created_ms) + { + return Ok(false); + } + self.remove(tenant, token).await?; + list.retain(|d| d.token != token); + Ok(true) + } + + /// What this tenant can be woken on, oldest first. + pub async fn devices(&self, tenant: [u8; 16]) -> Vec { + self.devices + .lock() + .await + .get(&tenant) + .cloned() + .unwrap_or_default() + } + + /// How many devices this tenant has enrolled. Never the tokens. + pub async fn count(&self, tenant: [u8; 16]) -> u32 { + self.devices + .lock() + .await + .get(&tenant) + .map(|l| l.len() as u32) + .unwrap_or(0) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use s3fs_core::backend::memory::MemoryBackend; + use s3fs_core::{Config, MasterSecret}; + + const ALICE: [u8; 16] = [1; 16]; + const BOB: [u8; 16] = [2; 16]; + + fn token(seed: &str) -> String { + format!("{seed}{}", "x".repeat(MIN_TOKEN_LEN)) + } + + async fn memory() -> Arc { + let backend = Arc::new(MemoryBackend::new()); + Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([5u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("filesystem") + } + + async fn registry() -> Arc { + DeviceRegistry::open(memory().await).await.unwrap() + } + + #[tokio::test] + async fn a_device_registered_by_one_tenant_is_invisible_to_another() { + let r = registry().await; + r.register(ALICE, &token("alice"), 1000).await.unwrap(); + assert_eq!(r.count(ALICE).await, 1); + assert_eq!(r.count(BOB).await, 0); + assert!(r.devices(BOB).await.is_empty()); + } + + #[tokio::test] + async fn registering_a_token_a_tenant_already_has_changes_nothing() { + let r = registry().await; + r.register(ALICE, &token("a"), 1000).await.unwrap(); + r.register(ALICE, &token("a"), 9999).await.unwrap(); + let devices = r.devices(ALICE).await; + assert_eq!(devices.len(), 1); + assert_eq!( + devices[0].created_ms, 1000, + "re-registering moved the device to the back of the eviction queue" + ); + } + + #[tokio::test] + async fn the_device_cap_evicts_the_oldest_rather_than_refusing_the_newest() { + let r = registry().await; + for i in 0..MAX_DEVICES_PER_TENANT { + r.register(ALICE, &token(&format!("d{i}")), 1000 + i as u64) + .await + .unwrap(); + } + r.register(ALICE, &token("newest"), 9000).await.unwrap(); + + let devices = r.devices(ALICE).await; + assert_eq!(devices.len(), MAX_DEVICES_PER_TENANT); + assert!( + devices.iter().any(|d| d.token == token("newest")), + "the newest device was refused instead of admitted" + ); + assert!( + !devices.iter().any(|d| d.token == token("d0")), + "the oldest device survived the cap" + ); + } + + /// Accepting an enrolment past the cap would write a store that the next + /// restart refuses to load — an enclave made unbootable by ordinary use. + #[tokio::test] + async fn an_enrolment_past_the_global_cap_is_refused_rather_than_written() { + let r = DeviceRegistry::open_with_limits( + memory().await, + DeviceLimits { + max_devices: 3, + per_tenant: 8, + }, + ) + .await + .unwrap(); + r.register(ALICE, &token("a"), 1).await.unwrap(); + r.register(BOB, &token("b"), 2).await.unwrap(); + r.register(BOB, &token("c"), 3).await.unwrap(); + + let err = r + .register(ALICE, &token("d"), 4) + .await + .unwrap_err() + .to_string(); + assert!(err.contains("full"), "unhelpful refusal: {err}"); + assert_eq!( + r.count(ALICE).await, + 1, + "a refused enrolment was still stored" + ); + + // Forgetting one makes room again. + r.forget(BOB, &token("c")).await.unwrap(); + r.register(ALICE, &token("d"), 5).await.unwrap(); + assert_eq!(r.count(ALICE).await, 2); + } + + /// A store can be over the cap legitimately — a lowered cap in a new image, + /// or records written before the cap was enforced on the way in. Booting is + /// what matters; these are cached addresses, not work. + #[tokio::test] + async fn a_store_over_its_cap_still_opens_and_drops_the_oldest() { + let fs = memory().await; + let roomy = DeviceRegistry::open_with_limits( + fs.clone(), + DeviceLimits { + max_devices: 8, + per_tenant: 8, + }, + ) + .await + .unwrap(); + for i in 0..6 { + roomy + .register(ALICE, &token(&format!("d{i}")), 1000 + i as u64) + .await + .unwrap(); + } + drop(roomy); + + // The same store, reopened under a tighter cap. + let tight = DeviceRegistry::open_with_limits( + fs, + DeviceLimits { + max_devices: 2, + per_tenant: 8, + }, + ) + .await + .expect("an over-full store must still open"); + let left = tight.devices(ALICE).await; + assert_eq!(left.len(), 2, "the store was not trimmed to its cap"); + assert_eq!( + left.iter().map(|d| d.created_ms).collect::>(), + vec![1004, 1005], + "trimming kept the wrong records" + ); + } + + #[tokio::test] + async fn forgetting_a_token_that_is_not_there_is_success_not_an_error() { + let r = registry().await; + r.forget(ALICE, &token("never")).await.unwrap(); + r.register(ALICE, &token("a"), 1).await.unwrap(); + r.forget(ALICE, &token("other")).await.unwrap(); + assert_eq!(r.count(ALICE).await, 1); + } + + #[tokio::test] + async fn a_token_that_is_not_a_plausible_registration_token_is_refused() { + let r = registry().await; + assert!(r.register(ALICE, "short", 1).await.is_err()); + assert!(r + .register(ALICE, &"x".repeat(MAX_TOKEN_LEN + 1), 1) + .await + .is_err()); + assert!( + r.register(ALICE, &format!("has spaces{}", "x".repeat(40)), 1) + .await + .is_err(), + "a token with a path-hostile character was accepted" + ); + assert_eq!(r.count(ALICE).await, 0); + } + + #[tokio::test] + async fn a_token_reregistered_while_a_prune_was_in_flight_survives_it() { + let r = registry().await; + r.register(ALICE, &token("a"), 1000).await.unwrap(); + // The forwarder failed against the record created at 1000; by the time + // it prunes, the client has re-enrolled and the record is newer. + r.forget(ALICE, &token("a")).await.unwrap(); + r.register(ALICE, &token("a"), 5000).await.unwrap(); + + assert!(!r + .forget_if_unchanged(ALICE, &token("a"), 1000) + .await + .unwrap()); + assert_eq!(r.count(ALICE).await, 1, "a live re-enrolment was pruned"); + + assert!(r + .forget_if_unchanged(ALICE, &token("a"), 5000) + .await + .unwrap()); + assert_eq!(r.count(ALICE).await, 0); + } + + #[tokio::test] + async fn a_published_device_record_survives_a_fresh_filesystem_mount() { + let backend = Arc::new(MemoryBackend::new()); + let master = MasterSecret::from_bytes([7u8; 32]); + let fs = Fs::create( + backend.clone(), + backend.clone(), + &master, + [8u8; 16], + Arc::new(Config::default()), + ) + .await + .unwrap(); + let r = DeviceRegistry::open(fs.clone()).await.unwrap(); + r.register(ALICE, &token("a"), 1000).await.unwrap(); + r.register(BOB, &token("b"), 2000).await.unwrap(); + drop(r); + drop(fs); + + let fs = Fs::mount( + backend.clone(), + backend, + &master, + [8u8; 16], + Arc::new(Config::default()), + None, + ) + .await + .unwrap(); + let reopened = DeviceRegistry::open(fs).await.unwrap(); + assert_eq!(reopened.count(ALICE).await, 1); + assert_eq!(reopened.count(BOB).await, 1); + assert_eq!(reopened.devices(ALICE).await[0].token, token("a")); + } + + #[tokio::test] + async fn an_unpublished_temporary_file_is_discarded_but_a_corrupt_record_stops_startup() { + let fs = memory().await; + let r = DeviceRegistry::open(fs.clone()).await.unwrap(); + r.register(ALICE, &token("a"), 1000).await.unwrap(); + + let dir = format!("/{RUNTIME_DIR}/{DEVICES_DIR}/{}", hex::encode(ALICE)); + let h = fs + .open(&format!("{dir}/orphan.tmp"), OpenFlags::create_new()) + .await + .unwrap(); + fs.close(&h).await.unwrap(); + assert_eq!( + DeviceRegistry::open(fs.clone()) + .await + .unwrap() + .count(ALICE) + .await, + 1, + "an unpublished temporary file was treated as a record" + ); + + // A published record that will not decode is a different matter: it is + // storage this runtime does not understand, and guessing is worse than + // refusing to start. + let h = fs + .open( + &format!("{dir}/{}", record_name(&token("bad"))), + OpenFlags::create_new(), + ) + .await + .unwrap(); + fs.pwrite(&h, 0, b"not json").await.unwrap(); + fs.close(&h).await.unwrap(); + assert!(DeviceRegistry::open(fs).await.is_err()); + } + + #[tokio::test] + async fn a_record_that_names_a_tenant_other_than_its_directory_is_refused() { + let fs = memory().await; + let r = DeviceRegistry::open(fs.clone()).await.unwrap(); + r.register(ALICE, &token("a"), 1000).await.unwrap(); + + // Alice's directory, a record claiming to be Bob's: the filename still + // matches its token, so only the directory check catches this. + let forged = StoredDevice { + version: RECORD_VERSION, + tenant: BOB, + token: token("forged"), + created_ms: 1, + }; + let path = format!( + "/{RUNTIME_DIR}/{DEVICES_DIR}/{}/{}", + hex::encode(ALICE), + record_name(&forged.token) + ); + let h = fs.open(&path, OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&h, 0, &serde_json::to_vec(&forged).unwrap()) + .await + .unwrap(); + fs.close(&h).await.unwrap(); + + assert!(DeviceRegistry::open(fs).await.is_err()); + } + + #[tokio::test] + async fn a_record_whose_hash_does_not_match_its_token_is_refused() { + let fs = memory().await; + let r = DeviceRegistry::open(fs.clone()).await.unwrap(); + r.register(ALICE, &token("a"), 1000).await.unwrap(); + + // The right owner, but filed under another token's name — so only the + // filename check catches this one. + let device = StoredDevice { + version: RECORD_VERSION, + tenant: ALICE, + token: token("real"), + created_ms: 1, + }; + let path = format!( + "/{RUNTIME_DIR}/{DEVICES_DIR}/{}/{}", + hex::encode(ALICE), + record_name(&token("someone-else")) + ); + let h = fs.open(&path, OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&h, 0, &serde_json::to_vec(&device).unwrap()) + .await + .unwrap(); + fs.close(&h).await.unwrap(); + + assert!(DeviceRegistry::open(fs).await.is_err()); + } + + #[tokio::test] + async fn a_record_from_an_unknown_version_is_refused_rather_than_misparsed() { + let fs = memory().await; + DeviceRegistry::open(fs.clone()).await.unwrap(); + + let device = StoredDevice { + version: RECORD_VERSION + 1, + tenant: ALICE, + token: token("a"), + created_ms: 1, + }; + let dir = ensure_dir( + &fs, + &fs.lookup_at(&fs.root(), RUNTIME_DIR).await.unwrap(), + DEVICES_DIR, + ) + .await + .unwrap(); + ensure_dir(&fs, &dir, &hex::encode(ALICE)).await.unwrap(); + let path = format!( + "/{RUNTIME_DIR}/{DEVICES_DIR}/{}/{}", + hex::encode(ALICE), + record_name(&device.token) + ); + let h = fs.open(&path, OpenFlags::create_new()).await.unwrap(); + fs.pwrite(&h, 0, &serde_json::to_vec(&device).unwrap()) + .await + .unwrap(); + fs.close(&h).await.unwrap(); + + assert!(DeviceRegistry::open(fs).await.is_err()); + } + + /// The registry sits above every tenant scope, so no guest can name it. + #[tokio::test] + async fn device_records_live_above_every_tenant_scope() { + let fs = memory().await; + let r = DeviceRegistry::open(fs.clone()).await.unwrap(); + r.register(ALICE, &token("a"), 1).await.unwrap(); + + let root = fs.root(); + let runtime = fs.lookup_at(&root, RUNTIME_DIR).await.unwrap(); + let devices = fs.lookup_at(&runtime, DEVICES_DIR).await.unwrap(); + assert!(fs.lookup_at(&devices, &hex::encode(ALICE)).await.is_ok()); + + // A tenant's scope is a sibling of `/runtime`, never an ancestor. + let tenant = crate::tenant::tenant_root_by_id(&fs, ALICE).await.unwrap(); + assert_ne!(tenant.scope.objid(), root.objid()); + assert_ne!(tenant.scope.objid(), devices.objid()); + } +} diff --git a/runtime/src/notify/fcm.rs b/runtime/src/notify/fcm.rs new file mode 100644 index 0000000..e58ab9a --- /dev/null +++ b/runtime/src/notify/fcm.rs @@ -0,0 +1,689 @@ +//! Talking to Firebase Cloud Messaging. +//! +//! # The message carries nothing +//! +//! Every request built here is **data-only**. There is no `notification` +//! object, no title and no body, and that absence is the security property the +//! whole feature rests on: the payload crosses the parent instance and then +//! Google, so anything put in it is disclosed to both. What travels is an +//! opaque category and an optional tenant-local reference — labels the guest +//! chose, meaningful only to an app that can already reach the enclave. +//! +//! It is also the correct shape mechanically. A `notification` block is +//! rendered by the OS without the app running, so a wake that carried one would +//! show text *and* fail to wake anything. +//! +//! # Errors are classified by what the service said +//! +//! Never by matching on a `Debug` string. FCM names its own failures, and the +//! three that matter are told apart by name: a token that is gone, a message +//! that will never be accepted, and everything else — which is worth retrying, +//! credential failures included, because those heal on the next refresh. + +use std::sync::Arc; +use std::time::Duration; + +use anyhow::{ensure, Result}; + +use super::oauth::{AccessToken, ServiceAccount, TokenResponse}; + +/// Where FCM lives, unless a deployment points somewhere else. +const DEFAULT_ENDPOINT: &str = "https://fcm.googleapis.com"; +/// A wake that arrives tomorrow is noise, not a notification. FCM's own default +/// is four weeks. +const TTL_SECS: u64 = 3600; +/// How long one exchange with FCM may take, end to end. +/// +/// Deliberately here rather than inside a transport: the forwarder sends to a +/// tenant's devices one after another, so an endpoint that accepts a connection +/// and then says nothing would stop *every* tenant's wakes for ever. Shorter +/// than the forwarder's flush deadline, so a shutdown can still finish inside +/// its own bound. +const EXCHANGE_TIMEOUT: Duration = Duration::from_secs(3); + +/// Labels are identifiers, not prose. The charset is what keeps them out of +/// trouble in JSON, in logs, and in whatever the app switches on. +const MAX_LABEL: usize = 64; + +/// How to reach FCM. +pub struct NotifyConfig { + pub project_id: String, + pub service_account: ServiceAccount, + /// For tests and local endpoints. A downgrade path: pointed at a plain + /// `http://` stub it hands wake signals to whatever is listening. PCR0 + /// records which was built, which is the only reason this is acceptable. + pub endpoint: Option, +} + +impl std::fmt::Debug for NotifyConfig { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("NotifyConfig") + .field("project_id", &self.project_id) + .field("service_account", &self.service_account) + .field("endpoint", &self.endpoint) + .finish() + } +} + +/// One HTTP exchange. The seam that keeps the network out of the tests. +#[async_trait::async_trait] +pub trait FcmTransport: Send + Sync + std::fmt::Debug { + async fn send( + &self, + request: http::Request>, + ) -> std::result::Result>, String>; +} + +/// What went wrong, in the only three shapes the caller acts on differently. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum SendError { + /// This registration token is gone. Prune it; never retry it. + DeadToken(String), + /// This message will never be accepted. Drop it and count it. + Refused(String), + /// Worth trying again, credential failures included. + Transient(String), +} + +impl std::fmt::Display for SendError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + SendError::DeadToken(m) => write!(f, "registration token is gone: {m}"), + SendError::Refused(m) => write!(f, "refused: {m}"), + SendError::Transient(m) => write!(f, "transient: {m}"), + } + } +} + +pub fn valid_label(label: &str) -> bool { + !label.is_empty() + && label.len() <= MAX_LABEL + && label + .bytes() + .all(|b| b.is_ascii_alphanumeric() || matches!(b, b'-' | b'_')) +} + +/// The body of a wake signal. +/// +/// Public so a test can assert on exactly what would go on the wire, which is +/// the only part of this module worth pinning byte for byte. +pub fn wake_message( + token: &str, + category: &str, + reference: Option<&str>, + now_ms: u64, +) -> serde_json::Value { + let mut data = serde_json::Map::new(); + // A schema version the app can switch on. Free now, impossible to add + // later without a release that understands both. + data.insert("v".into(), "1".into()); + data.insert("category".into(), category.into()); + if let Some(reference) = reference { + data.insert("ref".into(), reference.into()); + } + + serde_json::json!({ + "message": { + "token": token, + // Data-only. See the module docs: there is no `notification` key + // here, and adding one would disclose its contents to the parent + // instance and to Google. + "data": data, + // High priority so a data-only message is delivered in Doze. + // Android meters this, so a chatty guest is throttled by the + // platform rather than by us. + "android": { "priority": "high", "ttl": format!("{TTL_SECS}s") }, + // The only combination APNs accepts for a silent wake: a + // background push type with content-available, at priority 5. + // Priority 10 with a background type is rejected outright. + "apns": { + "headers": { + "apns-push-type": "background", + "apns-priority": "5", + "apns-expiration": (now_ms / 1000 + TTL_SECS).to_string(), + }, + "payload": { "aps": { "content-available": 1 } }, + }, + } + }) +} + +/// Reads FCM's own error name out of a response. +fn classify(status: u16, body: &[u8]) -> SendError { + let parsed: serde_json::Value = serde_json::from_slice(body).unwrap_or(serde_json::Value::Null); + let code = parsed["error"]["details"] + .as_array() + .and_then(|details| { + details + .iter() + .find_map(|d| d["errorCode"].as_str().map(str::to_string)) + }) + .or_else(|| parsed["error"]["status"].as_str().map(str::to_string)) + .unwrap_or_default(); + let message = parsed["error"]["message"] + .as_str() + .unwrap_or("no message") + .to_string(); + let detail = format!("{status} {code}: {message}"); + + match code.as_str() { + "UNREGISTERED" | "NOT_FOUND" => SendError::DeadToken(detail), + "INVALID_ARGUMENT" | "SENDER_ID_MISMATCH" | "THIRD_PARTY_AUTH_ERROR" => { + SendError::Refused(detail) + } + // Credentials stay transient on purpose: an expired token heals on the + // next refresh, and treating it as fatal turns routine rotation into + // an outage. + _ => SendError::Transient(detail), + } +} + +/// The FCM half of notify: mints access tokens and sends wake signals. +/// +/// Owned by the single forwarder task, so the token cache is a plain field +/// behind `&mut self` — there is no refresh race to design against. +#[derive(Debug)] +pub struct FcmClient { + config: NotifyConfig, + transport: Arc, + cached: Option, +} + +impl FcmClient { + pub fn new(config: NotifyConfig, transport: Arc) -> Self { + FcmClient { + config, + transport, + cached: None, + } + } + + /// Send one request, and never wait on it for ever. + async fn exchange( + &self, + request: http::Request>, + ) -> std::result::Result>, SendError> { + let uri = request.uri().to_string(); + match tokio::time::timeout(EXCHANGE_TIMEOUT, self.transport.send(request)).await { + Ok(Ok(response)) => Ok(response), + Ok(Err(e)) => Err(SendError::Transient(format!("reaching {uri}: {e}"))), + Err(_) => Err(SendError::Transient(format!( + "{uri} did not answer within {EXCHANGE_TIMEOUT:?}" + ))), + } + } + + fn endpoint(&self) -> &str { + self.config.endpoint.as_deref().unwrap_or(DEFAULT_ENDPOINT) + } + + /// Mint an access token, or reuse the one we have. + async fn access_token(&mut self, now_ms: u64) -> std::result::Result { + if let Some(token) = &self.cached { + if token.is_fresh(now_ms) { + return Ok(token.token.to_string()); + } + } + + let assertion = self + .config + .service_account + .assertion(now_ms) + .map_err(|e| SendError::Transient(format!("signing the assertion: {e:#}")))?; + let body = self.config.service_account.token_form(&assertion); + let request = http::Request::builder() + .method("POST") + .uri(&self.config.service_account.token_uri) + .header("content-type", "application/x-www-form-urlencoded") + .body(body.into_bytes()) + .map_err(|e| SendError::Transient(format!("building the token request: {e}")))?; + + let response = self.exchange(request).await?; + let status = response.status().as_u16(); + if status != 200 { + let body = String::from_utf8_lossy(response.body()).to_string(); + let parsed: serde_json::Value = + serde_json::from_slice(response.body()).unwrap_or(serde_json::Value::Null); + let detail = format!("the token endpoint answered {status}: {body}"); + // `invalid_client` and `unauthorized_client` name a credential that + // will never work, so the startup probe can refuse to boot on them. + // `invalid_grant` deliberately is not in that set: it is as likely + // to be a clock a minute out as a bad key, and an enclave that + // refused to start over clock skew would be worse than one that + // retries. See the note in `oauth`. + return Err(match parsed["error"].as_str().unwrap_or_default() { + "invalid_client" | "unauthorized_client" => SendError::Refused(detail), + _ => SendError::Transient(detail), + }); + } + + let parsed: TokenResponse = serde_json::from_slice(response.body()) + .map_err(|e| SendError::Transient(format!("decoding the token response: {e}")))?; + let token = AccessToken::new(parsed, now_ms); + let value = token.token.to_string(); + self.cached = Some(token); + Ok(value) + } + + /// Prove the credential works, without waking anybody. + /// + /// Minting a token needs exactly the permission sending needs, so this is a + /// real check rather than a reachability ping — and it disturbs no device. + pub async fn probe(&mut self, now_ms: u64) -> std::result::Result<(), SendError> { + self.access_token(now_ms).await.map(|_| ()) + } + + /// Send one wake signal to one device. + pub async fn wake( + &mut self, + token: &str, + category: &str, + reference: Option<&str>, + now_ms: u64, + ) -> std::result::Result<(), SendError> { + let access = self.access_token(now_ms).await?; + let body = serde_json::to_vec(&wake_message(token, category, reference, now_ms)) + .map_err(|e| SendError::Refused(format!("encoding the message: {e}")))?; + let uri = format!( + "{}/v1/projects/{}/messages:send", + self.endpoint(), + self.config.project_id + ); + + let send = |access: &str, body: &[u8]| { + http::Request::builder() + .method("POST") + .uri(&uri) + .header("authorization", format!("Bearer {access}")) + .header("content-type", "application/json; charset=utf-8") + .body(body.to_vec()) + }; + + let request = send(&access, &body) + .map_err(|e| SendError::Refused(format!("building the request: {e}")))?; + let response = self.exchange(request).await?; + + let status = response.status().as_u16(); + if status == 200 { + return Ok(()); + } + // One retry on 401, and exactly one: the cached token may simply have + // been revoked early. A second 401 is transient with backoff, never a + // refresh loop. + if status == 401 { + self.cached = None; + let access = self.access_token(now_ms).await?; + let request = send(&access, &body) + .map_err(|e| SendError::Refused(format!("building the request: {e}")))?; + let response = self.exchange(request).await?; + if response.status().as_u16() == 200 { + return Ok(()); + } + return Err(classify(response.status().as_u16(), response.body())); + } + Err(classify(status, response.body())) + } +} + +/// Validate what a guest supplied before any of it reaches a request. +pub fn check_labels(category: &str, reference: Option<&str>) -> Result<()> { + ensure!( + valid_label(category), + "a category is 1-{MAX_LABEL} characters of letters, digits, '-' or '_'. \ + It is a label, not a message: a wake signal carries no text." + ); + if let Some(reference) = reference { + ensure!( + valid_label(reference), + "a reference is 1-{MAX_LABEL} characters of letters, digits, '-' or '_'" + ); + } + Ok(()) +} + +/// Build the client the runtime actually uses. +pub fn client(config: NotifyConfig, transport: Arc) -> Result { + ensure!( + !config.project_id.trim().is_empty(), + "an FCM project id is required" + ); + Ok(FcmClient::new(config, transport)) +} + +#[cfg(test)] +pub(crate) mod testing { + use super::*; + use std::sync::Mutex; + + /// Records what was sent and answers from a script. + #[derive(Debug, Default)] + pub struct Recorder { + pub sent: Mutex>>>, + /// Answers, consumed in order. The last one repeats once exhausted. + pub replies: Mutex>, + /// When set, every send hangs — standing in for a push service that + /// has stopped answering. + pub stall: std::sync::atomic::AtomicBool, + } + + impl Recorder { + pub fn with(replies: Vec<(u16, &str)>) -> Arc { + Arc::new(Recorder { + sent: Mutex::new(Vec::new()), + replies: Mutex::new( + replies + .into_iter() + .map(|(s, b)| (s, b.to_string())) + .collect(), + ), + stall: std::sync::atomic::AtomicBool::new(false), + }) + } + + /// Every request that was not the token exchange. + pub fn messages(&self) -> Vec { + self.sent + .lock() + .unwrap() + .iter() + .filter(|r| r.uri().path().ends_with("messages:send")) + .map(|r| serde_json::from_slice(r.body()).expect("a JSON body")) + .collect() + } + + pub fn requests(&self) -> Vec>> { + self.sent + .lock() + .unwrap() + .iter() + .map(|r| { + let mut clone = http::Request::builder() + .method(r.method().clone()) + .uri(r.uri().clone()); + for (name, value) in r.headers() { + clone = clone.header(name, value); + } + clone.body(r.body().clone()).unwrap() + }) + .collect() + } + } + + #[async_trait::async_trait] + impl FcmTransport for Recorder { + async fn send( + &self, + request: http::Request>, + ) -> std::result::Result>, String> { + if self.stall.load(std::sync::atomic::Ordering::Relaxed) { + std::future::pending::<()>().await; + } + let token_request = request.uri().path().ends_with("/token"); + self.sent.lock().unwrap().push(request); + if token_request { + return Ok(http::Response::builder() + .status(200) + .body( + serde_json::json!({"access_token": "ya29.test", "expires_in": 3600}) + .to_string() + .into_bytes(), + ) + .unwrap()); + } + let mut replies = self.replies.lock().unwrap(); + let (status, body) = if replies.len() > 1 { + replies.remove(0) + } else { + replies.first().cloned().unwrap_or((200, "{}".to_string())) + }; + Ok(http::Response::builder() + .status(status) + .body(body.into_bytes()) + .unwrap()) + } + } +} + +#[cfg(test)] +mod tests { + use super::testing::Recorder; + use super::*; + + const TOKEN: &str = "cWWpFzVAQ0y7Zb3hJ8kLmN:APA91bHqRsTuVwXyZ0123456789abcdef"; + + fn account() -> ServiceAccount { + ServiceAccount::parse( + &serde_json::json!({ + "type": "service_account", + "project_id": "enclave-test", + "private_key_id": "kid-1", + "private_key": include_str!("testdata/service-account-key.pem"), + "client_email": "wake@enclave-test.iam.gserviceaccount.com", + }) + .to_string(), + ) + .unwrap() + } + + fn client_with(recorder: Arc) -> FcmClient { + FcmClient::new( + NotifyConfig { + project_id: "enclave-test".into(), + service_account: account(), + endpoint: None, + }, + recorder, + ) + } + + /// The claim the whole feature rests on. + #[tokio::test] + async fn a_wake_signal_is_data_only_and_carries_no_notification_block() { + let recorder = Recorder::with(vec![(200, "{}")]); + let mut client = client_with(recorder.clone()); + client + .wake(TOKEN, "approval-needed", Some("txn-9f2"), 1_000_000) + .await + .unwrap(); + + let message = &recorder.messages()[0]["message"]; + assert!( + message["notification"].is_null(), + "a wake signal carried a notification block: {message}" + ); + assert_eq!(message["data"]["category"], "approval-needed"); + assert_eq!(message["data"]["ref"], "txn-9f2"); + assert_eq!(message["data"]["v"], "1"); + + // Nothing beyond the labels and the fixed keys reached the wire. + let data = message["data"].as_object().unwrap(); + let mut keys: Vec<_> = data.keys().map(String::as_str).collect(); + keys.sort(); + assert_eq!(keys, ["category", "ref", "v"]); + } + + #[tokio::test] + async fn the_send_names_the_project_and_carries_a_bearer_token() { + let recorder = Recorder::with(vec![(200, "{}")]); + let mut client = client_with(recorder.clone()); + client.wake(TOKEN, "ping", None, 0).await.unwrap(); + + let sent = recorder.requests(); + let message = sent + .iter() + .find(|r| r.uri().path().ends_with("messages:send")) + .expect("a send"); + assert_eq!( + message.uri().to_string(), + "https://fcm.googleapis.com/v1/projects/enclave-test/messages:send" + ); + assert_eq!( + message.headers()["authorization"], + "Bearer ya29.test", + "the send did not carry the minted access token" + ); + assert_eq!(message.method(), http::Method::POST); + } + + #[tokio::test] + async fn an_ios_wake_is_a_background_push_and_an_android_wake_is_high_priority() { + let recorder = Recorder::with(vec![(200, "{}")]); + let mut client = client_with(recorder.clone()); + client.wake(TOKEN, "ping", None, 1_000_000).await.unwrap(); + + let message = &recorder.messages()[0]["message"]; + assert_eq!(message["android"]["priority"], "high"); + assert_eq!(message["apns"]["headers"]["apns-push-type"], "background"); + assert_eq!(message["apns"]["headers"]["apns-priority"], "5"); + assert_eq!(message["apns"]["payload"]["aps"]["content-available"], 1); + } + + #[tokio::test] + async fn a_wake_expires_rather_than_arriving_a_day_late() { + let recorder = Recorder::with(vec![(200, "{}")]); + let mut client = client_with(recorder.clone()); + client.wake(TOKEN, "ping", None, 1_000_000).await.unwrap(); + + let message = &recorder.messages()[0]["message"]; + assert_eq!(message["android"]["ttl"], "3600s"); + assert_eq!( + message["apns"]["headers"]["apns-expiration"], + (1000 + TTL_SECS).to_string() + ); + } + + #[test] + fn a_category_that_is_not_a_plain_label_never_reaches_the_wire() { + assert!(check_labels("approval-needed", Some("txn_1")).is_ok()); + assert!(check_labels("", None).is_err()); + assert!(check_labels(&"x".repeat(MAX_LABEL + 1), None).is_err()); + assert!( + check_labels("Approve $4,000 to Acme", None).is_err(), + "prose was accepted as a category" + ); + assert!(check_labels("ok", Some("a b")).is_err()); + } + + #[tokio::test] + async fn an_unregistered_token_is_pruned_rather_than_retried() { + let body = serde_json::json!({"error": { + "status": "NOT_FOUND", "message": "Requested entity was not found.", + "details": [{"@type": "type.googleapis.com/google.firebase.fcm.v1.FcmError", + "errorCode": "UNREGISTERED"}]}}) + .to_string(); + let recorder = Recorder::with(vec![(404, &body)]); + let mut client = client_with(recorder); + let err = client.wake(TOKEN, "ping", None, 0).await.unwrap_err(); + assert!( + matches!(err, SendError::DeadToken(_)), + "a gone token was classified {err:?}" + ); + } + + /// The forwarder sends to a tenant's devices one after another, so an + /// endpoint that accepts a connection and then says nothing would stop + /// every tenant's wakes for ever. It has to come back as an ordinary + /// transient failure instead. + #[tokio::test(start_paused = true)] + async fn an_endpoint_that_never_answers_is_transient_rather_than_a_hang() { + let recorder = Recorder::with(vec![(200, "{}")]); + recorder + .stall + .store(true, std::sync::atomic::Ordering::Relaxed); + let mut client = client_with(recorder); + + let err = client.wake(TOKEN, "ping", None, 0).await.unwrap_err(); + match err { + SendError::Transient(detail) => { + assert!(detail.contains("did not answer"), "unhelpful: {detail}") + } + other => panic!("a stalled endpoint was classified {other:?}"), + } + } + + #[tokio::test] + async fn a_malformed_message_is_dropped_rather_than_retried_for_ever() { + let body = serde_json::json!({"error": { + "status": "INVALID_ARGUMENT", "message": "Invalid value at 'message.token'", + "details": [{"errorCode": "INVALID_ARGUMENT"}]}}) + .to_string(); + let recorder = Recorder::with(vec![(400, &body)]); + let mut client = client_with(recorder); + assert!(matches!( + client.wake(TOKEN, "ping", None, 0).await.unwrap_err(), + SendError::Refused(_) + )); + } + + #[tokio::test] + async fn a_throttled_or_unavailable_send_is_transient() { + for (status, code) in [ + (429, "QUOTA_EXCEEDED"), + (503, "UNAVAILABLE"), + (500, "INTERNAL"), + ] { + let body = + serde_json::json!({"error": {"status": code, "message": "later", "details": [{"errorCode": code}]}}) + .to_string(); + let recorder = Recorder::with(vec![(status, &body)]); + let mut client = client_with(recorder); + assert!( + matches!( + client.wake(TOKEN, "ping", None, 0).await.unwrap_err(), + SendError::Transient(_) + ), + "{code} was not treated as worth retrying" + ); + } + } + + /// An expired access token heals on the next refresh, so it must not be + /// fatal — but it must not loop either. + #[tokio::test] + async fn an_unauthorized_send_is_retried_once_with_a_fresh_token() { + let recorder = Recorder::with(vec![(401, "{}"), (200, "{}")]); + let mut client = client_with(recorder.clone()); + client.wake(TOKEN, "ping", None, 0).await.unwrap(); + assert_eq!( + recorder.messages().len(), + 2, + "the send was not retried after a 401" + ); + + // And a second 401 gives up rather than refreshing for ever. + let recorder = Recorder::with(vec![(401, "{}")]); + let mut client = client_with(recorder.clone()); + assert!(matches!( + client.wake(TOKEN, "ping", None, 0).await.unwrap_err(), + SendError::Transient(_) + )); + assert_eq!(recorder.messages().len(), 2, "a 401 loop"); + } + + #[tokio::test] + async fn a_cached_access_token_is_minted_once_for_many_sends() { + let recorder = Recorder::with(vec![(200, "{}")]); + let mut client = client_with(recorder.clone()); + for _ in 0..3 { + client.wake(TOKEN, "ping", None, 1_000).await.unwrap(); + } + let tokens = recorder + .requests() + .iter() + .filter(|r| r.uri().path().ends_with("/token")) + .count(); + assert_eq!(tokens, 1, "the access token was minted per send"); + } + + #[tokio::test] + async fn the_probe_mints_a_token_and_wakes_nobody() { + let recorder = Recorder::with(vec![(200, "{}")]); + let mut client = client_with(recorder.clone()); + client.probe(0).await.unwrap(); + assert!( + recorder.messages().is_empty(), + "the startup probe sent a message to a real device" + ); + assert_eq!(recorder.requests().len(), 1); + } +} diff --git a/runtime/src/notify/mod.rs b/runtime/src/notify/mod.rs new file mode 100644 index 0000000..8f20bde --- /dev/null +++ b/runtime/src/notify/mod.rs @@ -0,0 +1,778 @@ +//! Waking a tenant's devices, without telling anyone what happened. +//! +//! A guest cannot reach its user between requests, and by default it cannot +//! reach the network at all — [`crate::serve::EgressPolicy`] refuses every +//! outgoing request a guest makes, and a deployment that opens any names only +//! specific origins, never a push service. So the runtime holds the push +//! credential and sends on the guest's behalf, and the guest gets a host import +//! instead of a socket. +//! +//! # A wake signal carries nothing +//! +//! The payload crosses the parent instance — the party this enclave exists to +//! exclude — and then Google. So it contains an opaque category, an optional +//! tenant-local reference, and nothing else: no title, no body, no +//! `notification` object. The app wakes and fetches the detail over the +//! attested channel, where the parent is excluded again. +//! +//! That is not a setting. A flag would be something a deployment turns on and a +//! guest then uses, at which point it stops being a property. + +use std::collections::{HashSet, VecDeque}; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::{Arc, Mutex}; +use std::time::Duration; + +use anyhow::{ensure, Context, Result}; +use tokio::sync::{mpsc, Notify}; +use wasmtime_wasi::HostWallClock; + +use crate::clock::WallClockAdapter; +use crate::state::State; + +pub mod device; +pub mod fcm; +pub mod oauth; +pub mod transport; + +pub use device::{DeviceRegistry, StoredDevice, MAX_DEVICES, MAX_DEVICES_PER_TENANT}; +pub use fcm::{FcmClient, FcmTransport, NotifyConfig, SendError}; +pub use oauth::{AccessToken, ServiceAccount, TokenResponse}; +pub use transport::{web_pki_client_config, HttpsTransport}; + +/// Wakes waiting to be sent. Beyond this the oldest are dropped: a wake is a +/// hint, and a backlog of stale hints is worth less than a fresh one. +const QUEUE_CAPACITY: usize = 1024; +/// Distinct categories one tenant may have in flight. Reaching it is a guest +/// raising more kinds of signal than it can possibly need at once. +const MAX_INFLIGHT_PER_TENANT: usize = 8; +/// Wakes held back for retry. Also dropped oldest-first. +const MAX_PENDING: usize = 64; +const MIN_BACKOFF: Duration = Duration::from_millis(500); +const MAX_BACKOFF: Duration = Duration::from_secs(60); +/// One line about drops per interval, never one per drop. +const DROP_REPORT_INTERVAL: Duration = Duration::from_secs(60); +/// How long shutdown waits for the queue to empty before abandoning it. +const FLUSH_DEADLINE: Duration = Duration::from_secs(5); + +/// How long the startup probe may take before the enclave stops waiting. +/// +/// A notification client must never be what decides whether the enclave binds +/// its listener. +pub const NOTIFY_STARTUP_PROBE_TIMEOUT: Duration = Duration::from_secs(10); + +/// The per-invocation authority to use notify, stamped by the serving path. +/// +/// The tenant comes from the executing instance, never from the caller, which +/// is why no function in the WIT takes a tenant id. +#[derive(Clone)] +pub struct NotifyContext { + pub notifier: Arc, + pub tenant: [u8; 16], + pub interactive: bool, +} + +impl std::fmt::Debug for NotifyContext { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("NotifyContext") + .field("interactive", &self.interactive) + .finish_non_exhaustive() + } +} + +#[derive(Default, Debug)] +struct Counters { + queued_dropped: AtomicU64, + pending_dropped: AtomicU64, + refused: AtomicU64, + pruned: AtomicU64, + sent: AtomicU64, + /// Loop iterations. Only a test reads this, and only to prove an idle + /// forwarder is not burning a core. + wakeups: AtomicU64, +} + +#[derive(Debug)] +struct Wake { + tenant: [u8; 16], + category: String, + reference: Option, + /// When this wake may next be attempted. Zero means immediately. + due_ms: u64, + attempts: u32, +} + +/// The guest-facing half: owns the registry and the queue, owns no network. +pub struct Notifier { + registry: Arc, + clock: Arc, + tx: mpsc::Sender, + /// `(tenant, category)` already queued. A wake is idempotent, so a second + /// one for a category still waiting *is* the one already waiting — and + /// coalescing is also what stops one tenant filling a shared queue with + /// repeats of a single signal. + /// + /// A `std` mutex, never tokio's: `raise` holds it, and `raise` must contain + /// nothing that can pend. + inflight: Mutex>, + counters: Arc, +} + +impl std::fmt::Debug for Notifier { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("Notifier") + .field("counters", &self.counters) + .finish_non_exhaustive() + } +} + +impl Notifier { + pub fn now_ms(&self) -> u64 { + self.clock.now().as_millis().min(u64::MAX as u128) as u64 + } + + /// Enrol a device for this tenant. + pub async fn register_device(&self, tenant: [u8; 16], token: &str) -> Result<()> { + let now = self.now_ms(); + self.registry.register(tenant, token, now).await + } + + pub async fn forget_device(&self, tenant: [u8; 16], token: &str) -> Result<()> { + self.registry.forget(tenant, token).await + } + + pub async fn device_count(&self, tenant: [u8; 16]) -> u32 { + self.registry.count(tenant).await + } + + /// How many times the forwarder has come round its loop. + pub fn wakeups(&self) -> u64 { + self.counters.wakeups.load(Ordering::Relaxed) + } + + /// Queue a wake signal. + /// + /// **Not async, and it must stay that way.** This runs inside a host call + /// that holds the tenant's only instance slot, so it touches neither the + /// network nor the filesystem — the same invariant, and the same reason, as + /// `CloudWatchLogSink::emit`. + /// + /// A full queue returns `Ok`: a drop is not the guest's fault and not + /// something it can act on, and telling it would invite a retry that makes + /// the queue worse. Exceeding the per-tenant cap *does* return `Err`, + /// because that one is actionable — the guest is raising more distinct + /// categories than it has reason to. + pub fn raise(&self, tenant: [u8; 16], category: &str, reference: Option<&str>) -> Result<()> { + fcm::check_labels(category, reference)?; + + let key = (tenant, category.to_string()); + { + let mut inflight = self.inflight.lock().expect("inflight poisoned"); + if inflight.contains(&key) { + return Ok(()); + } + let held = inflight.iter().filter(|(t, _)| *t == tenant).count(); + ensure!( + held < MAX_INFLIGHT_PER_TENANT, + "this tenant already has {MAX_INFLIGHT_PER_TENANT} wake signals waiting" + ); + inflight.insert(key.clone()); + } + + let wake = Wake { + tenant, + category: category.to_string(), + reference: reference.map(str::to_string), + due_ms: 0, + attempts: 0, + }; + if self.tx.try_send(wake).is_err() { + self.inflight + .lock() + .expect("inflight poisoned") + .remove(&key); + self.counters.queued_dropped.fetch_add(1, Ordering::Relaxed); + } + Ok(()) + } + + fn release(&self, tenant: [u8; 16], category: &str) { + self.inflight + .lock() + .expect("inflight poisoned") + .remove(&(tenant, category.to_string())); + } +} + +/// The task that actually talks to FCM. +pub struct NotifyForwarder { + task: tokio::task::JoinHandle<()>, + stop: Arc, +} + +impl NotifyForwarder { + /// Let the queue drain, then stop. Bounded: a push service having a bad day + /// must not hold the enclave open. + pub async fn shutdown(mut self) { + self.stop.notify_one(); + match tokio::time::timeout(FLUSH_DEADLINE, &mut self.task).await { + Ok(Ok(())) => {} + Ok(Err(e)) => tracing::warn!(error = %e, "the notification forwarder ended badly"), + Err(_) => { + tracing::warn!( + "the notification forwarder did not flush in time; dropping queued wakes" + ); + self.task.abort(); + } + } + } +} + +/// Build the guest-facing half and the task that drains it. +pub fn start( + registry: Arc, + clock: Arc, + client: FcmClient, +) -> (Arc, NotifyForwarder) { + let (tx, rx) = mpsc::channel::(QUEUE_CAPACITY); + let counters = Arc::new(Counters::default()); + let stop = Arc::new(Notify::new()); + let notifier = Arc::new(Notifier { + registry: registry.clone(), + clock: clock.clone(), + tx, + inflight: Mutex::new(HashSet::new()), + counters: counters.clone(), + }); + let task = tokio::spawn(forward( + rx, + client, + registry, + notifier.clone(), + counters, + clock, + stop.clone(), + )); + (notifier, NotifyForwarder { task, stop }) +} + +fn backoff_for(attempts: u32) -> Duration { + MIN_BACKOFF + .saturating_mul(1u32 << attempts.min(7)) + .min(MAX_BACKOFF) +} + +async fn forward( + mut rx: mpsc::Receiver, + mut client: FcmClient, + registry: Arc, + notifier: Arc, + counters: Arc, + clock: Arc, + stop: Arc, +) { + let mut pending: VecDeque = VecDeque::new(); + let mut last_report = tokio::time::Instant::now(); + + loop { + counters.wakeups.fetch_add(1, Ordering::Relaxed); + let now = clock.now().as_millis().min(u64::MAX as u128) as u64; + // Only ever computed from a deadline that has something behind it. A + // timer armed on an empty queue is how an idle forwarder burns a core. + let wait = pending + .iter() + .map(|w| w.due_ms.saturating_sub(now)) + .min() + .map(Duration::from_millis); + + tokio::select! { + biased; + + _ = stop.notified() => { + drain(&mut rx, &mut pending); + for wake in pending.drain(..) { + deliver(&mut client, ®istry, ¬ifier, &counters, wake, now).await; + } + report(&counters, true); + return; + } + + received = rx.recv() => { + match received { + Some(wake) => pending.push_back(wake), + // Every Notifier is gone; nothing can queue again. + None => { + for wake in pending.drain(..) { + deliver(&mut client, ®istry, ¬ifier, &counters, wake, now).await; + } + report(&counters, true); + return; + } + } + } + + _ = async { tokio::time::sleep(wait.unwrap_or_default()).await }, if wait.is_some() => {} + } + + while pending.len() > MAX_PENDING { + pending.pop_front(); + counters.pending_dropped.fetch_add(1, Ordering::Relaxed); + } + + let now = clock.now().as_millis().min(u64::MAX as u128) as u64; + let mut carry = VecDeque::new(); + while let Some(wake) = pending.pop_front() { + if wake.due_ms > now { + carry.push_back(wake); + continue; + } + if let Some(retry) = + deliver(&mut client, ®istry, ¬ifier, &counters, wake, now).await + { + carry.push_back(retry); + } + } + pending = carry; + + if last_report.elapsed() >= DROP_REPORT_INTERVAL { + report(&counters, false); + last_report = tokio::time::Instant::now(); + } + } +} + +fn drain(rx: &mut mpsc::Receiver, pending: &mut VecDeque) { + while let Ok(wake) = rx.try_recv() { + pending.push_back(wake); + } +} + +/// Send one wake to every device the tenant has. Returns the wake again if it +/// is worth another attempt. +async fn deliver( + client: &mut FcmClient, + registry: &Arc, + notifier: &Arc, + counters: &Arc, + mut wake: Wake, + now_ms: u64, +) -> Option { + // Released as the wake leaves the queue, so the next signal for this + // category can be raised while this one is still being retried. + notifier.release(wake.tenant, &wake.category); + + let devices = registry.devices(wake.tenant).await; + if devices.is_empty() { + return None; + } + + // Sequential, not fanned out: at most a handful of tokens over one reused + // connection is cheap, and it orders a token's pruning against the next + // send rather than racing it. + let mut retry = false; + for device in devices { + match client + .wake( + &device.token, + &wake.category, + wake.reference.as_deref(), + now_ms, + ) + .await + { + Ok(()) => { + counters.sent.fetch_add(1, Ordering::Relaxed); + } + Err(SendError::DeadToken(detail)) => { + counters.pruned.fetch_add(1, Ordering::Relaxed); + tracing::debug!(detail, "pruning a registration token FCM says is gone"); + if let Err(e) = registry + .forget_if_unchanged(wake.tenant, &device.token, device.created_ms) + .await + { + tracing::warn!(error = %e, "could not prune a dead registration token"); + } + } + Err(SendError::Refused(detail)) => { + counters.refused.fetch_add(1, Ordering::Relaxed); + tracing::warn!(detail, "FCM refused a wake signal"); + } + Err(SendError::Transient(detail)) => { + tracing::debug!(detail, "a wake signal will be retried"); + retry = true; + } + } + } + + if !retry { + return None; + } + wake.attempts += 1; + wake.due_ms = now_ms.saturating_add(backoff_for(wake.attempts).as_millis() as u64); + Some(wake) +} + +fn report(counters: &Arc, final_report: bool) { + let queued = counters.queued_dropped.swap(0, Ordering::Relaxed); + let pending = counters.pending_dropped.swap(0, Ordering::Relaxed); + let refused = counters.refused.swap(0, Ordering::Relaxed); + let pruned = counters.pruned.swap(0, Ordering::Relaxed); + let sent = counters.sent.swap(0, Ordering::Relaxed); + if queued + pending + refused + pruned + sent == 0 { + return; + } + // One line per interval. A warning per dropped wake would bury the console + // exactly when something is already going wrong. + tracing::info!( + sent, + dropped_queued = queued, + dropped_pending = pending, + refused, + pruned, + final_report, + "notification delivery" + ); +} + +fn context(state: &State, enrolment: bool) -> Result { + let ctx = state + .notify + .clone() + .context("notifications require an authenticated tenant and a configured sender")?; + // `wake` is deliberately absent from this check: it spends an enrolment an + // interactive call already made, and background work raising a wake is the + // whole point of the feature. + ensure!( + !enrolment || ctx.interactive, + "background work cannot enrol or remove a device" + ); + Ok(ctx) +} + +/// The ABI is defined in wit/notify/notify.wit. No function accepts a tenant id. +pub fn add_to_linker(linker: &mut wasmtime::component::Linker) -> wasmtime::Result<()> { + let mut notify = linker.instance("enclave:notify/notify@0.1.0")?; + + notify.func_wrap_async("register-device", |store, (token,): (String,)| { + Box::new(async move { + let result = async move { + let ctx = context(store.data(), true)?; + ctx.notifier.register_device(ctx.tenant, &token).await + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + })?; + + notify.func_wrap_async("forget-device", |store, (token,): (String,)| { + Box::new(async move { + let result = async move { + let ctx = context(store.data(), true)?; + ctx.notifier.forget_device(ctx.tenant, &token).await + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + })?; + + notify.func_wrap_async("devices", |store, (): ()| { + Box::new(async move { + let result: Result = async move { + let ctx = context(store.data(), false)?; + Ok(ctx.notifier.device_count(ctx.tenant).await) + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + })?; + + notify.func_wrap_async( + "wake", + |store, (category, reference): (String, Option)| { + Box::new(async move { + let result = async move { + let ctx = context(store.data(), false)?; + ctx.notifier + .raise(ctx.tenant, &category, reference.as_deref()) + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + }, + )?; + + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::fcm::testing::Recorder; + use super::*; + use crate::clock::HostClock; + use s3fs_core::backend::memory::MemoryBackend; + use s3fs_core::{Config, Fs, MasterSecret}; + + const ALICE: [u8; 16] = [1; 16]; + const BOB: [u8; 16] = [2; 16]; + + fn token(seed: &str) -> String { + format!("{seed}{}", "y".repeat(40)) + } + + fn account() -> ServiceAccount { + ServiceAccount::parse( + &serde_json::json!({ + "type": "service_account", + "project_id": "enclave-test", + "private_key_id": "kid-1", + "private_key": include_str!("testdata/service-account-key.pem"), + "client_email": "wake@enclave-test.iam.gserviceaccount.com", + }) + .to_string(), + ) + .unwrap() + } + + async fn fixture( + replies: Vec<(u16, &str)>, + ) -> ( + Arc, + NotifyForwarder, + Arc, + Arc, + ) { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([3u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .unwrap(); + let registry = DeviceRegistry::open(fs).await.unwrap(); + let clock = Arc::new(WallClockAdapter::new(Box::new(HostClock)).unwrap()); + let recorder = Recorder::with(replies); + let client = FcmClient::new( + NotifyConfig { + project_id: "enclave-test".into(), + service_account: account(), + endpoint: None, + }, + recorder.clone(), + ); + let (notifier, forwarder) = start(registry.clone(), clock, client); + (notifier, forwarder, registry, recorder) + } + + /// A `Notifier` with no forwarder behind it. + /// + /// `raise` is where coalescing and the per-tenant cap live, and both are + /// about what is *waiting*. With a forwarder running, a wake's key is + /// released the moment it is dequeued — before it is even sent — so an + /// assertion about what is in flight becomes a race with that task. Holding + /// the receiver and never reading it makes the queue stand still. + async fn detached() -> (Arc, mpsc::Receiver) { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([4u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .unwrap(); + let (tx, rx) = mpsc::channel::(QUEUE_CAPACITY); + let notifier = Arc::new(Notifier { + registry: DeviceRegistry::open(fs).await.unwrap(), + clock: Arc::new(WallClockAdapter::new(Box::new(HostClock)).unwrap()), + tx, + inflight: Mutex::new(HashSet::new()), + counters: Arc::new(Counters::default()), + }); + (notifier, rx) + } + + /// Wait for a condition the forwarder reaches asynchronously. + async fn until(mut done: impl FnMut() -> bool) { + for _ in 0..200 { + if done() { + return; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + panic!("the forwarder never got there"); + } + + #[tokio::test(flavor = "multi_thread")] + async fn a_wake_reaches_every_device_the_tenant_enrolled() { + let (notifier, forwarder, _registry, recorder) = fixture(vec![(200, "{}")]).await; + notifier.register_device(ALICE, &token("a")).await.unwrap(); + notifier.register_device(ALICE, &token("b")).await.unwrap(); + + notifier.raise(ALICE, "task-done", Some("job")).unwrap(); + until(|| recorder.messages().len() == 2).await; + + let sent: Vec<_> = recorder.messages(); + assert!(sent + .iter() + .all(|m| m["message"]["data"]["category"] == "task-done")); + forwarder.shutdown().await; + } + + /// The invariant that keeps a push service out of the request path. + #[tokio::test(flavor = "multi_thread")] + async fn raising_a_wake_never_delays_the_guest_call_that_raised_it() { + let (notifier, forwarder, _registry, recorder) = fixture(vec![(200, "{}")]).await; + recorder + .stall + .store(true, std::sync::atomic::Ordering::Relaxed); + notifier.register_device(ALICE, &token("a")).await.unwrap(); + + let started = std::time::Instant::now(); + for i in 0..QUEUE_CAPACITY * 2 { + // Distinct categories, so nothing is coalesced away and the queue + // genuinely fills. + let _ = notifier.raise(ALICE, &format!("c{i}"), None); + } + let elapsed = started.elapsed(); + assert!( + elapsed < Duration::from_secs(2), + "raising wakes against a stalled push service took {elapsed:?}" + ); + forwarder.shutdown().await; + } + + #[tokio::test] + async fn repeated_wakes_for_one_category_coalesce_into_one_send() { + let (notifier, mut rx) = detached().await; + for _ in 0..50 { + notifier.raise(ALICE, "task-done", None).unwrap(); + } + assert_eq!(notifier.inflight.lock().unwrap().len(), 1); + + // And one reached the queue, not fifty. + let mut queued = 0; + while rx.try_recv().is_ok() { + queued += 1; + } + assert_eq!(queued, 1, "a repeated wake was queued {queued} times"); + } + + #[tokio::test] + async fn one_tenant_cannot_hold_more_than_its_share_of_wakes_in_flight() { + let (notifier, _rx) = detached().await; + for i in 0..MAX_INFLIGHT_PER_TENANT { + notifier.raise(ALICE, &format!("c{i}"), None).unwrap(); + } + assert!( + notifier.raise(ALICE, "one-too-many", None).is_err(), + "a tenant queued more distinct categories than its share" + ); + // The cap is per tenant, not global: Bob is unaffected. + notifier.raise(BOB, "mine", None).unwrap(); + } + + #[tokio::test(flavor = "multi_thread")] + async fn a_dead_token_is_pruned_and_the_tenants_other_devices_still_get_woken() { + let gone = serde_json::json!({"error": { + "status": "NOT_FOUND", "message": "gone", + "details": [{"errorCode": "UNREGISTERED"}]}}) + .to_string(); + // The first device answers UNREGISTERED, the second accepts. + let (notifier, forwarder, registry, recorder) = + fixture(vec![(404, &gone), (200, "{}")]).await; + notifier + .register_device(ALICE, &token("dead")) + .await + .unwrap(); + notifier + .register_device(ALICE, &token("live")) + .await + .unwrap(); + + notifier.raise(ALICE, "task-done", None).unwrap(); + until(|| recorder.messages().len() == 2).await; + // The prune lands after the send it was decided by. + for _ in 0..200 { + if registry.devices(ALICE).await.len() == 1 { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + + // The dead one is gone, the live one remains. + let left = registry.devices(ALICE).await; + assert_eq!(left.len(), 1, "the dead token was not pruned: {left:?}"); + assert_eq!(left[0].token, token("live")); + forwarder.shutdown().await; + } + + #[tokio::test(flavor = "multi_thread")] + async fn a_wake_for_a_tenant_with_no_devices_is_dropped_quietly() { + let (notifier, forwarder, _registry, recorder) = fixture(vec![(200, "{}")]).await; + notifier.raise(ALICE, "nobody-home", None).unwrap(); + tokio::time::sleep(Duration::from_millis(100)).await; + assert!(recorder.messages().is_empty()); + forwarder.shutdown().await; + } + + #[tokio::test(flavor = "multi_thread")] + async fn a_category_that_is_not_a_label_is_refused_before_it_is_queued() { + let (notifier, forwarder, _registry, recorder) = fixture(vec![(200, "{}")]).await; + notifier.register_device(ALICE, &token("a")).await.unwrap(); + assert!(notifier + .raise(ALICE, "Approve $4,000 to Acme", None) + .is_err()); + tokio::time::sleep(Duration::from_millis(50)).await; + assert!( + recorder.messages().is_empty(), + "prose reached the wire as a category" + ); + forwarder.shutdown().await; + } + + /// Five minutes of virtual time with nothing to send is a handful of + /// ticks, not thousands of iterations. + #[tokio::test(start_paused = true)] + async fn an_idle_forwarder_does_not_spin() { + let (notifier, _forwarder, _registry, _recorder) = fixture(vec![(200, "{}")]).await; + let before = notifier.wakeups(); + tokio::time::sleep(Duration::from_secs(300)).await; + let woke = notifier.wakeups() - before; + assert!( + woke < 50, + "the forwarder woke {woke} times while idle over five minutes" + ); + } + + #[tokio::test(flavor = "multi_thread")] + async fn shutdown_flushes_what_is_queued_and_cannot_hold_the_enclave_open() { + let (notifier, forwarder, _registry, recorder) = fixture(vec![(200, "{}")]).await; + notifier.register_device(ALICE, &token("a")).await.unwrap(); + notifier.raise(ALICE, "last-word", None).unwrap(); + forwarder.shutdown().await; + assert_eq!( + recorder.messages().len(), + 1, + "a queued wake was lost at shutdown" + ); + + // And a stalled service cannot hold shutdown open past the deadline. + let (notifier, forwarder, _registry, recorder) = fixture(vec![(200, "{}")]).await; + recorder + .stall + .store(true, std::sync::atomic::Ordering::Relaxed); + notifier.register_device(ALICE, &token("a")).await.unwrap(); + notifier.raise(ALICE, "never-lands", None).unwrap(); + let started = std::time::Instant::now(); + forwarder.shutdown().await; + assert!( + started.elapsed() < FLUSH_DEADLINE + Duration::from_secs(2), + "shutdown waited {:?} on a stalled push service", + started.elapsed() + ); + } +} diff --git a/runtime/src/notify/oauth.rs b/runtime/src/notify/oauth.rs new file mode 100644 index 0000000..3759b13 --- /dev/null +++ b/runtime/src/notify/oauth.rs @@ -0,0 +1,392 @@ +//! Proving to Google that this enclave may send for its Firebase project. +//! +//! FCM's HTTP v1 API takes an OAuth2 access token, and obtaining one means +//! signing a JWT with a service account's private key and exchanging it at +//! Google's token endpoint. There is no long-lived API key to present instead: +//! the legacy server-key API was decommissioned in 2024. +//! +//! # What is here and what is not +//! +//! This module signs and parses. It opens no sockets — the exchange itself is +//! [`super::fcm`]'s, so everything below can be tested without a network, and +//! the signature can be checked against the key that produced it rather than +//! against a recorded blob. +//! +//! # The clock +//! +//! `iat` and `exp` come from the runtime's trusted clock, which in an enclave +//! is PTP. This is the one place an outside party checks that clock: a skew +//! beyond Google's tolerance comes back as `invalid_grant`, which is +//! indistinguishable from a credential that was never valid. If tokens stop +//! minting after an image change, suspect the clock before the key. + +use anyhow::{ensure, Context, Result}; +use aws_lc_rs::rand::SystemRandom; +use aws_lc_rs::signature::{RsaKeyPair, RSA_PKCS1_SHA256}; +use base64::Engine as _; +use serde::Deserialize; +use zeroize::Zeroizing; + +/// The only scope this runtime asks for. Sending is all it does. +const SCOPE: &str = "https://www.googleapis.com/auth/firebase.messaging"; +/// Where a service account exchanges an assertion, unless its JSON says +/// otherwise. Whatever this ends up being is also the assertion's `aud`, and +/// Google refuses the pair if they disagree. +const DEFAULT_TOKEN_URI: &str = "https://oauth2.googleapis.com/token"; +/// Google refuses an assertion that claims longer than an hour. +const ASSERTION_LIFETIME_SECS: u64 = 3600; +/// Replace a token this long before it expires, so a send never races one out. +pub const REFRESH_SKEW_MS: u64 = 60_000; +/// An access token is never treated as living longer than this, whatever the +/// response said. Trusting a number from the network to decide when to next +/// talk to the network is how a stale token becomes a permanent one. +const MAX_LIFETIME_MS: u64 = 3600 * 1000; + +/// Percent-encoded so the form body needs no encoder: the value is a constant, +/// and the only other field is a JWT, which is base64url and already safe. +const GRANT_TYPE: &str = "urn%3Aietf%3Aparams%3Aoauth%3Agrant-type%3Ajwt-bearer"; + +fn b64(bytes: &[u8]) -> String { + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(bytes) +} + +/// The fields of a Google service-account JSON this runtime uses. +#[derive(Deserialize)] +struct ServiceAccountJson { + #[serde(rename = "type")] + kind: String, + project_id: String, + private_key_id: String, + private_key: String, + client_email: String, + #[serde(default)] + token_uri: Option, +} + +/// A parsed, usable service account. +/// +/// It cannot exist in an invalid state: the key is decoded and accepted by +/// `aws-lc-rs` before this is constructed, so a malformed credential fails at +/// boot rather than on the first wake somebody was waiting for. +pub struct ServiceAccount { + pub project_id: String, + pub client_email: String, + pub token_uri: String, + private_key_id: String, + key: RsaKeyPair, +} + +impl std::fmt::Debug for ServiceAccount { + /// Names the account, never the key. `ServeConfig` derives `Debug` and is + /// logged at startup, so this is on a path that reaches the console. + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("ServiceAccount") + .field("project_id", &self.project_id) + .field("client_email", &self.client_email) + .finish_non_exhaustive() + } +} + +impl ServiceAccount { + pub fn parse(json: &str) -> Result { + let parsed: ServiceAccountJson = + serde_json::from_str(json).context("the FCM service account is not valid JSON")?; + ensure!( + parsed.kind == "service_account", + "expected a service-account key, got {:?}", + parsed.kind + ); + ensure!( + !parsed.project_id.trim().is_empty() + && !parsed.client_email.trim().is_empty() + && !parsed.private_key_id.trim().is_empty(), + "the FCM service account is missing project_id, client_email or private_key_id" + ); + + let block = pem::parse(parsed.private_key.as_bytes()) + .context("the service account's private_key is not valid PEM")?; + ensure!( + block.tag() == "PRIVATE KEY", + "expected a PKCS#8 \"PRIVATE KEY\" block, found {:?}. A key exported as \ + \"RSA PRIVATE KEY\" is PKCS#1 and has to be converted.", + block.tag() + ); + let key = RsaKeyPair::from_pkcs8(block.contents()) + .map_err(|e| anyhow::anyhow!("the service account's private key was rejected: {e}"))?; + + Ok(ServiceAccount { + project_id: parsed.project_id, + client_email: parsed.client_email, + token_uri: parsed + .token_uri + .filter(|u| !u.trim().is_empty()) + .unwrap_or_else(|| DEFAULT_TOKEN_URI.to_string()), + private_key_id: parsed.private_key_id, + key, + }) + } + + /// The signed assertion Google exchanges for an access token. + pub fn assertion(&self, now_ms: u64) -> Result { + let now = now_ms / 1000; + let header = serde_json::json!({ + "alg": "RS256", + "typ": "JWT", + // Which of the account's keys signed this, so rotation does not + // need both sides to change at the same instant. + "kid": self.private_key_id, + }); + let claims = serde_json::json!({ + "iss": self.client_email, + "scope": SCOPE, + "aud": self.token_uri, + "iat": now, + "exp": now + ASSERTION_LIFETIME_SECS, + }); + + let signing_input = format!( + "{}.{}", + b64(&serde_json::to_vec(&header)?), + b64(&serde_json::to_vec(&claims)?) + ); + let mut signature = vec![0u8; self.key.public_modulus_len()]; + self.key + .sign( + &RSA_PKCS1_SHA256, + &SystemRandom::new(), + signing_input.as_bytes(), + &mut signature, + ) + .map_err(|e| anyhow::anyhow!("signing the FCM assertion: {e}"))?; + + Ok(format!("{signing_input}.{}", b64(&signature))) + } + + /// The `application/x-www-form-urlencoded` body of the token request. + pub fn token_form(&self, assertion: &str) -> String { + format!("grant_type={GRANT_TYPE}&assertion={assertion}") + } +} + +/// What Google sends back. +#[derive(Deserialize)] +pub struct TokenResponse { + pub access_token: String, + pub expires_in: u64, +} + +/// An access token, and when to stop using it. +pub struct AccessToken { + pub token: Zeroizing, + pub expires_at_ms: u64, +} + +impl std::fmt::Debug for AccessToken { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("AccessToken") + .field("expires_at_ms", &self.expires_at_ms) + .finish_non_exhaustive() + } +} + +impl AccessToken { + pub fn new(response: TokenResponse, now_ms: u64) -> Self { + let lifetime = response + .expires_in + .saturating_mul(1000) + .min(MAX_LIFETIME_MS); + AccessToken { + token: Zeroizing::new(response.access_token), + expires_at_ms: now_ms.saturating_add(lifetime), + } + } + + /// Whether this token is still worth sending with. + pub fn is_fresh(&self, now_ms: u64) -> bool { + now_ms.saturating_add(REFRESH_SKEW_MS) < self.expires_at_ms + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const KEY: &str = include_str!("testdata/service-account-key.pem"); + + fn account_json() -> String { + serde_json::json!({ + "type": "service_account", + "project_id": "enclave-test", + "private_key_id": "kid-1", + "private_key": KEY, + "client_email": "wake@enclave-test.iam.gserviceaccount.com", + "token_uri": "https://oauth2.googleapis.com/token", + }) + .to_string() + } + + fn account() -> ServiceAccount { + ServiceAccount::parse(&account_json()).expect("the fixture parses") + } + + fn part(jwt: &str, index: usize) -> serde_json::Value { + let raw = jwt.split('.').nth(index).expect("a JWT part"); + let bytes = base64::engine::general_purpose::URL_SAFE_NO_PAD + .decode(raw) + .expect("base64url"); + serde_json::from_slice(&bytes).expect("JSON") + } + + #[test] + fn the_assertion_carries_the_issuer_scope_and_audience_google_requires() { + let a = account(); + let jwt = a.assertion(1_000_000).unwrap(); + + let header = part(&jwt, 0); + assert_eq!(header["alg"], "RS256"); + assert_eq!(header["kid"], "kid-1"); + + let claims = part(&jwt, 1); + assert_eq!(claims["iss"], "wake@enclave-test.iam.gserviceaccount.com"); + assert_eq!(claims["scope"], SCOPE); + assert_eq!( + claims["aud"], "https://oauth2.googleapis.com/token", + "aud must equal the endpoint the assertion is posted to, or Google refuses it" + ); + assert_eq!(claims["iat"], 1000); + assert_eq!(claims["exp"], 1000 + ASSERTION_LIFETIME_SECS); + } + + /// Verified against the key that signed it, rather than compared to a + /// recorded blob: that proves the signing is real, where a fixed string + /// would only prove it is unchanged. + #[test] + fn the_assertion_verifies_against_the_service_accounts_own_public_key() { + use aws_lc_rs::signature::{KeyPair, UnparsedPublicKey, RSA_PKCS1_2048_8192_SHA256}; + + let a = account(); + let jwt = a.assertion(1_000_000).unwrap(); + let (signing_input, signature) = jwt.rsplit_once('.').expect("three parts"); + let signature = base64::engine::general_purpose::URL_SAFE_NO_PAD + .decode(signature) + .unwrap(); + + let public = UnparsedPublicKey::new( + &RSA_PKCS1_2048_8192_SHA256, + a.key.public_key().as_ref().to_vec(), + ); + public + .verify(signing_input.as_bytes(), &signature) + .expect("the assertion does not verify under its own key"); + + // And a tampered claim set does not. + let forged = format!("{signing_input}x"); + assert!(public.verify(forged.as_bytes(), &signature).is_err()); + } + + #[test] + fn a_service_account_missing_a_required_field_is_refused_before_any_network_call() { + for missing in ["project_id", "client_email", "private_key_id"] { + let mut value: serde_json::Value = serde_json::from_str(&account_json()).unwrap(); + value[missing] = serde_json::json!(""); + assert!( + ServiceAccount::parse(&value.to_string()).is_err(), + "an empty {missing} was accepted" + ); + } + let mut value: serde_json::Value = serde_json::from_str(&account_json()).unwrap(); + value["type"] = serde_json::json!("authorized_user"); + assert!( + ServiceAccount::parse(&value.to_string()).is_err(), + "a user credential was accepted as a service account" + ); + } + + #[test] + fn a_private_key_that_is_not_pkcs8_is_refused_at_startup() { + let mut value: serde_json::Value = serde_json::from_str(&account_json()).unwrap(); + value["private_key"] = serde_json::json!( + "-----BEGIN PRIVATE KEY-----\nbm90IGEga2V5\n-----END PRIVATE KEY-----\n" + ); + assert!(ServiceAccount::parse(&value.to_string()).is_err()); + + // PKCS#1, the other thing an export commonly produces. Refused by tag, + // with a message that says what to do about it. + let mut value: serde_json::Value = serde_json::from_str(&account_json()).unwrap(); + value["private_key"] = serde_json::json!(KEY.replace("PRIVATE KEY", "RSA PRIVATE KEY")); + let err = ServiceAccount::parse(&value.to_string()) + .unwrap_err() + .to_string(); + assert!(err.contains("PKCS#1"), "unhelpful error: {err}"); + } + + #[test] + fn the_service_account_never_appears_in_a_debug_line() { + let a = account(); + let rendered = format!("{a:?}"); + assert!(rendered.contains("enclave-test")); + assert!( + !rendered.contains("PRIVATE KEY") && !rendered.contains("kid-1"), + "a debug line carried key material: {rendered}" + ); + + let token = AccessToken::new( + TokenResponse { + access_token: "ya29.secret-value".into(), + expires_in: 3600, + }, + 0, + ); + assert!( + !format!("{token:?}").contains("secret-value"), + "a debug line carried an access token" + ); + } + + #[test] + fn a_cached_access_token_is_reused_until_it_is_about_to_expire() { + let token = AccessToken::new( + TokenResponse { + access_token: "ya29.x".into(), + expires_in: 3600, + }, + 1_000_000, + ); + assert_eq!(token.expires_at_ms, 1_000_000 + 3_600_000); + assert!(token.is_fresh(1_000_000)); + assert!(token.is_fresh(1_000_000 + 3_600_000 - REFRESH_SKEW_MS - 1)); + assert!( + !token.is_fresh(1_000_000 + 3_600_000 - REFRESH_SKEW_MS), + "a token inside the refresh skew was still considered fresh" + ); + assert!(!token.is_fresh(9_999_999_999)); + } + + /// A lifetime from the network decides when we next talk to the network, + /// so it is clamped rather than believed. + #[test] + fn an_implausible_token_lifetime_is_clamped_rather_than_trusted() { + let token = AccessToken::new( + TokenResponse { + access_token: "ya29.x".into(), + expires_in: u64::MAX, + }, + 0, + ); + assert_eq!(token.expires_at_ms, MAX_LIFETIME_MS); + } + + #[test] + fn the_token_request_is_a_jwt_bearer_grant() { + let a = account(); + let jwt = a.assertion(0).unwrap(); + let body = a.token_form(&jwt); + assert!(body.starts_with( + "grant_type=urn%3Aietf%3Aparams%3Aoauth%3Agrant-type%3Ajwt-bearer&assertion=" + )); + assert!( + !body[body.find("assertion=").unwrap()..].contains('+'), + "the assertion needed escaping the body does not do" + ); + } +} diff --git a/runtime/src/notify/testdata/README.md b/runtime/src/notify/testdata/README.md new file mode 100644 index 0000000..913c655 --- /dev/null +++ b/runtime/src/notify/testdata/README.md @@ -0,0 +1,6 @@ +A throwaway RSA key, generated for the `notify::oauth` tests and used nowhere +else. It signs assertions to a token endpoint that only exists in those tests. + +It is committed for the same reason `deploy/qemu-nitro/pebble/key.pem` is: a +test that generates a 2048-bit key on every run pays for it on every run, and a +fixture makes the tests deterministic. Nothing outside `#[cfg(test)]` reads it. diff --git a/runtime/src/notify/testdata/service-account-key.pem b/runtime/src/notify/testdata/service-account-key.pem new file mode 100644 index 0000000..db4b7ce --- /dev/null +++ b/runtime/src/notify/testdata/service-account-key.pem @@ -0,0 +1,28 @@ +-----BEGIN PRIVATE KEY----- +MIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQCvw2xq8t9LTXUH +cyeYu7ZfDpvMoXMRXOtXTMczYrU6/L6XLFWacxeSdyCeEhzHLASDaR6UIQaVv4ic +achgEJ8MCitwXf99VJ1h5IXFatdZJdGoI9FifWAfN/FKutFgPjO49FSdvJo8fG4u +Zr1hJ201LKrH4j9DvG1aSTk+5Q94f3p+ZHPJ2FkFjAicNFPprlVTm6H7Sct7VWYc +oRXkBMx4TxyhWm+AHZlFzKJ6ZIR8SLC+Ieu7+5YVsE3xvUBVMvrLcGpFl/ioMbTY +HMTFooefdYbKaelUaJciv9lZnWKXhyvYPuYZwcC9sk09jeIWi9OlaBpDosPt6K70 +NQ2gWe5VAgMBAAECggEAHD/DTOQoteRm3xHnxxFCdEA3k7nOMff2fkNBj/V5Ldgh +9M+kGY0OeJSreiRsmiltt0Y9qy6srYRJe2Q4F5KMUYXP6gE9l0HygqGVS3/KyVH+ +ArFxDYyblqDp5+ojTT3qF7uzXt/JhVe1aMFMBlGtKHr7nuEy7FrcU4LJz91GcYX9 +Guj2oJJNwvBmoDR2w+4iJ7UZeYvpuqJ1qIs/YZVaE7dzFgyIbA5AR2s2/F+/OVDd +ePsqPDLOFHEyyhfYswMYaHaCMhixZkx6+y1i++lx0GIb2Thjb61TBWu/jHDsTB43 +twI/V2NIcmYQmUrXODPwbW/+kFBlz+Cf5863dLzAKQKBgQDq0Lq/BmRzLY+u3UDo +fYd27bz1rJif2DEmXonAWrfxTBJIoQrFKX9SsyzPJPbnKACuHB4jTVETHhjccKJ3 +FGHz1k986NaGo6iJ5UBin+Td3rIOzNrp781PZRBsLJW8P/siO14EZzFKlssFdkva +ZMnCkQlWOBb2/42zLvqMwxZaPQKBgQC/ntX4BiBWtJ0I6hHtjHEwMmZh0wMFBF8G +8zpgUCySClnjaoK6kLA9hwy4zDPDOtIZOieHYs+Rg16EjsEMeCS52gmrpdpEE8ob +eCKQNViEPOXm3pJBsuIlVrodaWO4Oo1Pvhc8LZBTZ63qvXV0/pKzfnWAjS88Vwc0 +8nxUlWRd+QKBgB3tpq+sP+dSOkr+VkSLo1VsLbZeXkGZS4JpcEM9DM7LdFUfeYDx +rhG7Vo28V1/VAGkwmkLDmv7FykNmc76bsXRjr1PrVVRpzZRtzMwFNyV0OdubDpfc +gZ2J8xLmh9sriHWvfWcwQ98O4yd6EWbvi6up0rfThFHM9qGM7lA8mT+9AoGAZHUr +C8p6bbpmkWPVXkpAlNn3XtW3QYwXHZeqRRADLdULZvRR8Okl3DvO6Zr0kCdoOh2I +16tv0oOiq7ADeTwLVPwAEeLzWLlfPaNvy1aMP1eF19Fbr+HOOXEMRZsY0l6v8txf +ZgclIPS78tK8n0dPNZbYlzptRx8BAjsV/2oKolECgYA0sPm2WbpwEApgeMc+nXbP +tbmPLZcEpXSm9Z3UdM0GnkY3DBGGRXyH1At3KlRIgBq4zN8ojNKn8cbSgKJ/a8aT +Ly6fCf18shOUq3jww1TZOa1TkUZFWNfdTjTIMXEzH/dvIlJtlJjWmStp+qrlBKYE +GBs4U6iXoioFnUaCwPT65w== +-----END PRIVATE KEY----- diff --git a/runtime/src/notify/transport.rs b/runtime/src/notify/transport.rs new file mode 100644 index 0000000..34c78c1 --- /dev/null +++ b/runtime/src/notify/transport.rs @@ -0,0 +1,111 @@ +//! The one place this runtime opens an outbound connection of its own. +//! +//! Everything else that leaves the enclave goes through an AWS SDK. This does +//! not, because Firebase is not AWS — so the trust decision is made here, in +//! code, rather than inherited from a feature name. +//! +//! # Why the roots are compiled in +//! +//! [`crate::net`] notes that the **parent instance answers DNS**. Certificate +//! validation is therefore the only thing standing between +//! `fcm.googleapis.com` and whatever the parent would prefer to point it at. +//! The anchor set is `webpki-roots`, compiled into the image and covered by +//! PCR0, rather than a file the image happens to ship — so what this enclave +//! trusts outbound is part of what a client attests to. +//! +//! There is deliberately **no custom certificate verifier** on this path. +//! `src/bin/passkey-client.rs` has one; that binary exists to drive a +//! self-signed harness and is behind the `testing` feature. Nothing here may +//! acquire one, whatever a test would find convenient. + +use anyhow::{Context, Result}; +use http_body_util::{BodyExt, Full}; +use hyper_util::client::legacy::Client; +use hyper_util::rt::TokioExecutor; + +use super::fcm::FcmTransport; + +/// Refuse a response larger than this rather than buffering whatever arrives. +/// FCM's answers are small; anything this size is a sign the far end is not +/// FCM. +const MAX_RESPONSE: usize = 64 * 1024; + +/// A hyper client pinned to the public roots. +pub struct HttpsTransport { + client: Client< + hyper_rustls::HttpsConnector, + Full, + >, +} + +impl std::fmt::Debug for HttpsTransport { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("HttpsTransport").finish_non_exhaustive() + } +} + +/// The runtime's one client TLS configuration: aws-lc-rs, safe protocol versions, and the +/// `webpki-roots` anchors compiled into the image. Shared with guest egress (`serve::egress`) so +/// there is still one root store to audit, not two. +pub fn web_pki_client_config() -> Result { + let mut roots = rustls::RootCertStore::empty(); + roots.extend(webpki_roots::TLS_SERVER_ROOTS.iter().cloned()); + Ok(rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .context("selecting TLS protocol versions for the FCM client")? + .with_root_certificates(roots) + .with_no_client_auth()) +} + +impl HttpsTransport { + /// `allow_plaintext` exists for the emulator harness, which points the + /// runtime at a local stub. It is the same downgrade `--guest-log-endpoint` + /// already is, and PCR0 records which image was built. + pub fn new(allow_plaintext: bool) -> Result { + let builder = + hyper_rustls::HttpsConnectorBuilder::new().with_tls_config(web_pki_client_config()?); + let connector = if allow_plaintext { + builder + .https_or_http() + .enable_http1() + .enable_http2() + .build() + } else { + builder.https_only().enable_http1().enable_http2().build() + }; + Ok(HttpsTransport { + client: Client::builder(TokioExecutor::new()).build(connector), + }) + } +} + +#[async_trait::async_trait] +impl FcmTransport for HttpsTransport { + async fn send( + &self, + request: http::Request>, + ) -> std::result::Result>, String> { + let (parts, body) = request.into_parts(); + let request = http::Request::from_parts(parts, Full::new(bytes::Bytes::from(body))); + + let response = self + .client + .request(request) + .await + .map_err(|e| format!("{e}"))?; + let (parts, body) = response.into_parts(); + + // Bounded: a body is read to answer a question, not to be stored, and + // an unbounded read is an allocation somebody else decides the size of. + let collected = http_body_util::Limited::new(body, MAX_RESPONSE) + .collect() + .await + .map_err(|e| format!("reading the response body: {e}"))?; + Ok(http::Response::from_parts( + parts, + collected.to_bytes().to_vec(), + )) + } +} diff --git a/runtime/src/random.rs b/runtime/src/random.rs new file mode 100644 index 0000000..7f0c214 --- /dev/null +++ b/runtime/src/random.rs @@ -0,0 +1,324 @@ +//! Where the guest's random bytes come from. +//! +//! `wasi:random/random` is what a guest builds keys, nonces and session +//! identifiers from. Inside an enclave those bytes should come from the Nitro +//! Security Module — the same root of trust that signs attestation documents — +//! rather than from a kernel pool that *is* NSM-seeded but says nothing about +//! it and fails silently if it isn't. +//! +//! Bytes come **straight from the device on every call**. There is no DRBG in +//! between, so the trust argument is "the NSM produced these" with nothing +//! else to reason about. The cost is real: the device answers 256 bytes per +//! ioctl, so a large request is a loop of them. +//! +//! ## Why this panics where the clock does not +//! +//! [`cap_rand::RngCore`] has no fallible method. A device failure must become +//! either a panic or silently degraded bytes, and this chooses the panic. +//! +//! That looks inconsistent beside [`crate::clock`], which serves a stale +//! reading rather than failing — so the difference is worth stating. A slightly +//! stale timestamp is still a timestamp, and a caller can notice. Predictable +//! bytes handed to a guest that believes them random are indistinguishable from +//! good ones at the point of use; the guest builds a key and nothing downstream +//! can ever detect it. Stopping is the only safe answer. + +use std::fmt; +use std::path::Path; +use std::sync::Arc; + +use anyhow::{Context, Result}; + +use nitro_nsm::{Nsm, NsmDevice}; + +/// Which entropy source to use. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum RandomSource { + /// NSM if the device opens, kernel otherwise. + #[default] + Auto, + /// NSM, or refuse to start. + Nsm, + /// The kernel's `getrandom(2)`. + Host, +} + +impl RandomSource { + pub fn parse(s: &str) -> Result { + match s.trim().to_ascii_lowercase().as_str() { + "auto" => Ok(RandomSource::Auto), + "nsm" => Ok(RandomSource::Nsm), + "host" => Ok(RandomSource::Host), + other => Err(format!("expected one of auto, nsm, host; got {other:?}")), + } + } +} + +/// The kernel's `getrandom(2)`, for development and CI. +#[derive(Debug, Default)] +pub struct HostEntropy; + +impl Nsm for HostEntropy { + fn get_random(&self, buf: &mut [u8]) -> Result<()> { + getrandom::fill(buf).map_err(|e| anyhow::anyhow!("kernel getrandom(2): {e}")) + } + + /// PCRs are a property of a real enclave. There is nothing to report, + /// extend or lock, so all three refuse rather than inventing a register + /// file that would let boot logic appear to work outside an enclave. + fn describe_pcr(&self, _index: u16) -> Result { + anyhow::bail!("no PCRs: entropy is coming from the kernel, not an NSM") + } + + fn extend_pcr(&self, _index: u16, _data: &[u8]) -> Result> { + anyhow::bail!("no PCRs to extend: entropy is coming from the kernel, not an NSM") + } + + fn lock_pcr(&self, _index: u16) -> Result<()> { + anyhow::bail!("no PCRs to lock: entropy is coming from the kernel, not an NSM") + } + + /// There is no attestation without an NSM, and no substitute worth + /// inventing. + /// + /// A document is a signature by AWS over what this enclave is running. On + /// a developer's machine there is nothing to sign it with and nothing true + /// to say, so this fails rather than returning a plausible-looking + /// structure — a caller that got one would be building the very confusion + /// attestation exists to prevent. + fn attest(&self, _request: &nitro_nsm::AttestationRequest) -> Result> { + anyhow::bail!( + "no attestation available: entropy is coming from the kernel, not an NSM. \ + Attestation needs a real enclave (or the QEMU nitro-enclave harness)." + ) + } + + fn describe(&self) -> String { + "kernel getrandom(2) (not NSM)".to_string() + } +} + +/// Resolve a source into an entropy provider, saying which one was chosen. +/// +/// A fallback here is reported at `error`, not `warn`. Running an enclave on +/// host entropy is not a degraded mode to note and move on from — every key the +/// guest generates afterwards rests on it. +pub fn open_entropy(source: RandomSource, device: &Path) -> Result> { + match source { + RandomSource::Host => { + tracing::warn!( + entropy = "host", + "using kernel entropy; not the Nitro Security Module" + ); + Ok(Arc::new(HostEntropy)) + } + RandomSource::Nsm => { + let nsm = NsmDevice::open(device) + .context("entropy source 'nsm' was required but the device is unusable")?; + tracing::info!(entropy = %nsm.describe(), "entropy source"); + Ok(Arc::new(nsm)) + } + RandomSource::Auto => match NsmDevice::open(device) { + Ok(nsm) => { + tracing::info!(entropy = %nsm.describe(), "entropy source"); + Ok(Arc::new(nsm)) + } + Err(e) => { + tracing::error!( + device = %device.display(), + error = format!("{e:#}"), + "NSM unavailable, falling back to kernel entropy; \ + inside an enclave this is a security failure, not a warning" + ); + Ok(Arc::new(HostEntropy)) + } + }, + } +} + +/// Serves `wasi:random/random` from an entropy source. +pub struct GuestRandom { + source: Arc, +} + +impl GuestRandom { + pub fn new(source: Arc) -> Self { + GuestRandom { source } + } + + /// A seed for `wasi:random/insecure-seed`, drawn once at startup. That + /// interface is a hash seed, not key material, so one draw is enough. + pub fn insecure_seed(&self) -> Result { + let mut bytes = [0u8; 16]; + self.source.get_random(&mut bytes)?; + Ok(u128::from_le_bytes(bytes)) + } +} + +impl fmt::Debug for GuestRandom { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("GuestRandom") + .field("source", &self.source) + .finish() + } +} + +impl cap_rand::RngCore for GuestRandom { + fn next_u32(&mut self) -> u32 { + let mut b = [0u8; 4]; + self.fill_bytes(&mut b); + u32::from_le_bytes(b) + } + + fn next_u64(&mut self) -> u64 { + let mut b = [0u8; 8]; + self.fill_bytes(&mut b); + u64::from_le_bytes(b) + } + + fn fill_bytes(&mut self, dest: &mut [u8]) { + // See the module docs: there is no way to report this, and continuing + // would hand the guest bytes it would treat as secret. + self.source + .get_random(dest) + .expect("entropy source failed; refusing to hand the guest predictable bytes"); + } + + fn try_fill_bytes(&mut self, dest: &mut [u8]) -> Result<(), cap_rand::Error> { + match self.source.get_random(dest) { + Ok(()) => Ok(()), + Err(e) => { + tracing::error!(error = format!("{e:#}"), "entropy source failed"); + Err(cap_rand::Error::from( + std::num::NonZeroU32::new(cap_rand::Error::CUSTOM_START) + .expect("CUSTOM_START is nonzero"), + )) + } + } + } +} + +/// `cap_rand::CryptoRng` is a marker promising the output is suitable for +/// cryptography. That holds: the bytes are the NSM's, or the kernel CSPRNG's. +impl cap_rand::CryptoRng for GuestRandom {} + +#[cfg(test)] +mod tests { + use super::*; + use cap_rand::RngCore; + use nitro_nsm::fake::FakeNsm; + use std::sync::atomic::Ordering; + + fn fake() -> Arc { + Arc::new(FakeNsm::new()) + } + + #[test] + fn source_parses_the_documented_values() { + assert_eq!(RandomSource::parse("auto"), Ok(RandomSource::Auto)); + assert_eq!(RandomSource::parse("NSM"), Ok(RandomSource::Nsm)); + assert_eq!(RandomSource::parse(" host "), Ok(RandomSource::Host)); + assert_eq!(RandomSource::default(), RandomSource::Auto); + assert!(RandomSource::parse("urandom").is_err()); + } + + #[test] + fn bytes_come_from_the_source() { + let nsm = fake(); + let mut rng = GuestRandom::new(nsm.clone()); + let mut buf = [0u8; 64]; + rng.fill_bytes(&mut buf); + + assert!(buf.iter().any(|&b| b != 0)); + assert!( + nsm.calls.load(Ordering::SeqCst) > 0, + "the device was not consulted" + ); + } + + /// Every draw must hit the device: that is the whole point of choosing + /// direct-from-NSM over a seeded generator. + #[test] + fn every_draw_consults_the_device() { + let nsm = fake(); + let mut rng = GuestRandom::new(nsm.clone()); + for _ in 0..5 { + let _ = rng.next_u64(); + } + assert_eq!(nsm.calls.load(Ordering::SeqCst), 5); + } + + #[test] + fn successive_draws_differ() { + let nsm = fake(); + let mut rng = GuestRandom::new(nsm); + let (mut a, mut b) = ([0u8; 32], [0u8; 32]); + rng.fill_bytes(&mut a); + rng.fill_bytes(&mut b); + assert_ne!(a, b); + } + + #[test] + fn a_large_request_is_filled_completely() { + let nsm = fake(); + let mut rng = GuestRandom::new(nsm.clone()); + let mut buf = vec![0u8; 4096]; + rng.fill_bytes(&mut buf); + + assert_eq!( + nsm.calls.load(Ordering::SeqCst), + 16, + "4096 / 256 byte chunks" + ); + assert!(!buf.iter().all(|&b| b == buf[0]), "output is constant"); + } + + /// A guest must never receive predictable bytes believing them random, so + /// a dead entropy source stops the process rather than degrading. + #[test] + #[should_panic(expected = "entropy source failed")] + fn a_dead_source_panics_rather_than_degrading() { + let nsm = fake(); + nsm.empty.store(true, Ordering::SeqCst); + let mut rng = GuestRandom::new(nsm); + rng.fill_bytes(&mut [0u8; 16]); + } + + #[test] + fn the_fallible_path_reports_rather_than_panicking() { + let nsm = fake(); + nsm.empty.store(true, Ordering::SeqCst); + let mut rng = GuestRandom::new(nsm); + assert!(rng.try_fill_bytes(&mut [0u8; 16]).is_err()); + } + + #[test] + fn the_insecure_seed_is_drawn_from_the_source() { + let nsm = fake(); + let rng = GuestRandom::new(nsm.clone()); + let seed = rng.insecure_seed().unwrap(); + assert_ne!(seed, 0); + assert!(nsm.calls.load(Ordering::SeqCst) > 0); + } + + #[test] + fn the_host_source_produces_usable_entropy() { + let host = HostEntropy; + let mut a = [0u8; 32]; + let mut b = [0u8; 32]; + host.get_random(&mut a).unwrap(); + host.get_random(&mut b).unwrap(); + assert_ne!(a, b); + assert!(a.iter().any(|&x| x != 0)); + assert!(host.describe().contains("not NSM")); + } + + /// `auto` degrades; `nsm` refuses. Same asymmetry as the clock. + #[test] + fn auto_falls_back_but_nsm_does_not() { + let missing = Path::new("/dev/definitely-not-nsm"); + let fallback = open_entropy(RandomSource::Auto, missing).expect("auto must fall back"); + assert!(fallback.describe().contains("not NSM")); + assert!(open_entropy(RandomSource::Nsm, missing).is_err()); + } +} diff --git a/runtime/src/run.rs b/runtime/src/run.rs new file mode 100644 index 0000000..ff68613 --- /dev/null +++ b/runtime/src/run.rs @@ -0,0 +1,197 @@ +//! Reading a guest component, and the environment every instance is built from. + +use std::path::Path; +use std::sync::Arc; + +use anyhow::{Context, Result}; +use s3fs_core::Fs; +use wasmtime_wasi::WasiCtxBuilder; + +use crate::clock::{TrustedClock, WallClockAdapter}; +use crate::guest_io::GuestLogs; +use crate::random::GuestRandom; +use crate::state::State; +use nitro_nsm::Nsm; + +/// Exit code for a failure before the guest ever started — the filesystem +/// would not mount, or the component would not compile. +pub const EXIT_RUNTIME_FAILURE: i32 = 71; + +/// How the runtime ended. +/// +/// The parent instance reads the enclave's exit status, and "it stopped +/// serving" and "the filesystem refused to mount" call for different +/// responses. The second is a security event, not a bug in the guest. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum GuestOutcome { + /// The runtime did what it was asked and stopped. + Success, + /// It stopped for a reason it could report. + Failed, +} + +impl GuestOutcome { + pub fn exit_code(self) -> i32 { + match self { + GuestOutcome::Success => 0, + GuestOutcome::Failed => 1, + } + } + + pub fn is_success(self) -> bool { + self.exit_code() == 0 + } +} + +/// Read a component from disk. +/// +/// Names the path it looked at on failure: inside an enclave image that is the +/// single most likely misconfiguration, and there is no shell to go and check. +pub fn read_component(path: &Path) -> Result> { + std::fs::read(path).with_context(|| format!("reading guest component at {}", path.display())) +} + +/// Everything a guest instance is built from, in a form that can be used more +/// than once. +/// +/// A `wasi:cli/command` guest needs one instance and one [`State`]. A +/// `wasi:http/proxy` guest needs a fresh `State` per request — a `Store` cannot +/// be reused across requests, and reusing one would leak a guest's resource +/// table into the next caller's request. Both must produce *identical* +/// environments, so both go through here rather than each building a +/// `WasiCtxBuilder` and drifting. +/// +/// The filesystem is shared, deliberately: `Arc` is one mounted +/// filesystem with one transaction stream, so writes from separate requests +/// land in the same store rather than racing two views of it. +pub struct GuestEnvironment { + fs: Arc, + clock: Arc, + entropy: Arc, + env: Vec<(String, String)>, + args: Vec, + /// Where the guest's stdout and stderr go. Shared by every instance this + /// environment builds, so however many guests run, their output meets one + /// bounded queue and one collector. + logs: GuestLogs, +} + +impl GuestEnvironment { + /// `logs` is required rather than defaulted, so every caller has to answer + /// the question of where guest output goes. There is no arrangement in + /// which it is inherited by omission — that was the previous behaviour, + /// and it put untrusted guest text straight onto the enclave console with + /// nothing marking it as untrusted. + pub fn new( + fs: Arc, + clock: Box, + entropy: Arc, + env: &[(String, String)], + args: &[String], + logs: GuestLogs, + ) -> Result { + Ok(GuestEnvironment { + fs, + // One adapter shared by the guest's clock interface and the + // filesystem's "now", so `set-times` cannot disagree with + // `wall-clock`. + clock: Arc::new(WallClockAdapter::new(clock)?), + entropy, + env: env.to_vec(), + args: args.to_vec(), + logs, + }) + } + + pub fn fs(&self) -> &Arc { + &self.fs + } + + pub fn clock(&self) -> &Arc { + &self.clock + } + + /// Build a fresh [`State`] whose guest sees `scope` as `/`. + /// + /// Everything else — the filesystem, the clock, the entropy source, the + /// environment, the arguments — is shared. One mounted filesystem, one + /// block cache, one transaction stream; what differs between two clients + /// is only which directory each of them calls `/`, and that separation is + /// enforced by the resolver rather than by anything the guest does. + pub fn new_state_scoped(&self, scope: Arc) -> Result { + let mut wasi = WasiCtxBuilder::new(); + // Closed, not inherited. An enclave has no console to read from, so an + // inherited stdin offered a guest nothing but a handle on whatever the + // parent had attached to this process. Stated rather than left to the + // builder's default, because "the guest cannot read stdin" is a + // decision and not an accident. + wasi.stdin(tokio::io::empty()); + // Never `inherit_stdio`. Guest output is untrusted, attacker-chosen + // text; on the enclave's own stdout it would be indistinguishable from + // the runtime's log lines. These carry it into `crate::guest_io` + // instead, tagged by stream and marked as guest-produced. + wasi.stdout(self.logs.stdout()); + wasi.stderr(self.logs.stderr()); + wasi.envs(&self.env); + wasi.args(&self.args); + wasi.wall_clock(SharedWallClock(self.clock.clone())); + + let random = GuestRandom::new(self.entropy.clone()); + wasi.insecure_random_seed(random.insecure_seed()?); + wasi.secure_random(random); + + Ok(State::scoped( + wasi.build(), + self.fs.clone(), + scope, + self.clock.clone(), + )) + } + + /// Build a fresh [`State`] with this environment's capabilities. + /// + /// Each call draws a new insecure-random seed from the entropy source, so + /// two guest instances do not share a hash seed. + pub fn new_state(&self) -> Result { + self.new_state_scoped(self.fs.root()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// A failure before the guest ever started is not the same as the runtime + /// stopping, and the parent reads the difference off the exit status. + #[test] + fn exit_codes_distinguish_the_outcomes() { + assert_eq!(GuestOutcome::Success.exit_code(), 0); + assert_eq!(GuestOutcome::Failed.exit_code(), 1); + assert!(GuestOutcome::Success.is_success()); + assert!(!GuestOutcome::Failed.is_success()); + assert_ne!(EXIT_RUNTIME_FAILURE, GuestOutcome::Failed.exit_code()); + } + + #[test] + fn a_missing_component_names_the_path_it_looked_at() { + let err = read_component(Path::new("/enclave/definitely-absent.wasm")).unwrap_err(); + assert!( + format!("{err:#}").contains("/enclave/definitely-absent.wasm"), + "error must name the path: {err:#}" + ); + } +} + +/// `WasiCtxBuilder::wall_clock` takes ownership, but the filesystem needs the +/// same clock for `set-times`. This shares one. +#[derive(Debug, Clone)] +pub struct SharedWallClock(pub Arc); + +impl wasmtime_wasi::HostWallClock for SharedWallClock { + fn resolution(&self) -> std::time::Duration { + self.0.resolution() + } + fn now(&self) -> std::time::Duration { + self.0.now() + } +} diff --git a/runtime/src/serve/acme.rs b/runtime/src/serve/acme.rs new file mode 100644 index 0000000..8b88a35 --- /dev/null +++ b/runtime/src/serve/acme.rs @@ -0,0 +1,724 @@ +//! Let's Encrypt certificates, obtained and kept inside the enclave. +//! +//! A self-signed certificate is enough for a client that verifies attestation +//! — the binding proves more than a CA signature ever could. It is not enough +//! for a browser, which cannot check attestations and will simply refuse. So +//! the enclave gets a real one, and the private key still never leaves it. +//! +//! ## Why TLS-ALPN-01 +//! +//! The challenge arrives on **port 443**, the one the parent already forwards +//! inbound. HTTP-01 would need port 80 forwarded as well, and DNS-01 would +//! need DNS credentials inside the enclave — a standing secret that could +//! mint certificates for the whole zone. +//! +//! ## What this does not prove +//! +//! A publicly trusted certificate says a CA agreed the operator controls the +//! domain. The operator *does* control the domain, and could obtain a second +//! certificate for it outside the enclave and terminate TLS themselves. The +//! attestation binding is what closes that: `user_data` names the certificate +//! this enclave is serving, so a client that checks it will not accept the +//! operator's. Let's Encrypt buys browser compatibility, not trust. +//! +//! ## Where the key is kept +//! +//! rustls-acme hands its cache the account key and the certificate's private +//! key as PEM. Both are sealed with AES-256-GCM under a key derived from the +//! master secret and written as an ordinary object, so the parent instance +//! stores ciphertext it cannot read. +//! +//! Not in the filesystem: [`crate::wasi::host_preopens`] gives the guest a +//! descriptor for the filesystem *root*, so anything kept there is readable by +//! the guest — and a guest holding the TLS private key could impersonate the +//! enclave to every client. +//! +//! Caching is not an optimisation. Let's Encrypt allows five duplicate +//! certificates per week; an enclave that re-issued on every boot would run +//! out and be unable to serve. + +use std::sync::{Arc, RwLock}; + +use crate::serve::tls::TlsIdentity; +use anyhow::{Context, Result}; +use s3fs_core::backend::{Backend, PutBlobInput}; +use s3fs_core::crypto::KeyMaterial; +use s3fs_core::FsError; + +/// The identity currently being served — a rustls config and the leaf that +/// config presents, together. +/// +/// A shared slot rather than a fixed value because ACME issues asynchronously +/// and renews later: the certificate at boot is not necessarily the one in use +/// an hour on. +/// +/// **The pair is the point.** A connection loads one identity and gets both the +/// configuration it hands to rustls and the certificate that configuration will +/// present, indivisibly. Holding only the leaf — as this did — meant the +/// certificate a connection was attested with came from a global read at +/// response time, which during a renewal is a different certificate from the +/// one the handshake used. There is no rustls API to ask a live connection +/// which certificate it was served, so the answer has to be arranged rather +/// than discovered. +#[derive(Debug, Clone, Default)] +pub struct CertificateSlot(Arc>>>); + +impl CertificateSlot { + /// A slot holding an identity that will never change. + pub fn fixed(identity: Arc) -> Self { + CertificateSlot(Arc::new(RwLock::new(Some(identity)))) + } + + pub fn empty() -> Self { + CertificateSlot::default() + } + + pub fn set(&self, identity: Arc) { + let changed = { + let mut slot = self.0.write().expect("certificate slot poisoned"); + let changed = slot.as_ref().map(|i| i.certificate_der.as_slice()) + != Some(identity.certificate_der.as_slice()); + *slot = Some(identity); + changed + }; + if changed { + if let Some(der) = self.leaf() { + tracing::info!( + certificate_sha256 = %hex::encode(nitro_attestation::sha256(&der)), + "serving certificate changed; new connections will use it" + ); + } + } + } + + /// The identity a connection should be served with. Load this **once** per + /// connection and keep it: reading again later can give a different answer. + pub fn get(&self) -> Option> { + self.0.read().expect("certificate slot poisoned").clone() + } + + /// The leaf of whatever is current, for reporting only. + pub fn leaf(&self) -> Option> { + self.get().map(|i| i.certificate_der.clone()) + } +} + +/// Object key for the sealed ACME material. +fn cache_key(prefix: &str, kind: &str, scope: &[String], directory_url: &str) -> String { + // The directory URL is part of the key: staging and production issue + // different certificates, and a cached staging certificate served in + // production would be rejected by every client. + let mut hasher = blake3::Hasher::new(); + hasher.update(directory_url.as_bytes()); + for item in scope { + hasher.update(b"\0"); + hasher.update(item.as_bytes()); + } + let digest = hasher.finalize(); + format!( + "{prefix}acme/{kind}-{}", + hex::encode(&digest.as_bytes()[..16]) + ) +} + +/// rustls-acme cache backed by the object store, sealed under the master key. +pub struct SealedAcmeCache { + backend: Arc, + prefix: String, + keys: Arc, + /// Updated whenever a certificate is loaded or stored. + certificate: CertificateSlot, +} + +impl std::fmt::Debug for SealedAcmeCache { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("SealedAcmeCache") + .field("prefix", &self.prefix) + .finish_non_exhaustive() + } +} + +impl SealedAcmeCache { + pub fn new( + backend: Arc, + prefix: String, + keys: Arc, + certificate: CertificateSlot, + ) -> Self { + SealedAcmeCache { + backend, + prefix, + keys, + certificate, + } + } + + async fn load(&self, key: &str) -> Result>> { + match self.backend.get_blob(key, None).await { + Ok(output) => { + let plaintext = unseal(&self.keys, key, &output.body) + .with_context(|| format!("unsealing {key}"))?; + Ok(Some(plaintext)) + } + // A first boot has no cache. Everything else is a real failure and + // must not be mistaken for one. + Err(FsError::NotFound) => Ok(None), + Err(e) => { + Err(anyhow::Error::msg(e.to_string())).with_context(|| format!("reading {key}")) + } + } + } + + async fn store(&self, key: &str, plaintext: &[u8]) -> Result<()> { + let sealed = seal(&self.keys, key, plaintext).with_context(|| format!("sealing {key}"))?; + self.backend + .put_blob(PutBlobInput::new(key, sealed.into())) + .await + .map_err(|e| anyhow::Error::msg(e.to_string())) + .with_context(|| format!("writing {key}"))?; + Ok(()) + } + + /// Pull the leaf certificate out of rustls-acme's PEM blob. + /// + /// The layout is fixed by `AcmeState::parse_cert`: the PKCS#8 private key + /// first, then the chain, leaf first. So the leaf is the second PEM block — + /// and the first must never be published, which is why the whole blob is + /// sealed. + fn publish_leaf(&self, pem: &[u8]) { + match identity_from_pem(pem) { + Ok(identity) => self.certificate.set(Arc::new(identity)), + Err(e) => tracing::error!( + error = %e, + "could not build a serving identity from the ACME certificate; \ + connections will keep using the previous one" + ), + } + } +} + +/// Build a complete serving identity from rustls-acme's cached blob. +/// +/// The layout is fixed by `AcmeState::parse_cert`: the PKCS#8 private key +/// first, then the chain, leaf first. Taking both halves here is what lets the +/// runtime serve from its own configuration rather than from rustls-acme's +/// resolver — which matters because the resolver is updated before the cache +/// is written, so anything reading the cache would lag what is being served. +/// Serving from the same value we attest removes the divergence rather than +/// narrowing it. +pub fn identity_from_pem(pem: &[u8]) -> Result { + let blocks = pem_blocks(pem)?; + let (key, chain) = blocks.split_first().context("ACME blob is empty")?; + if chain.is_empty() { + anyhow::bail!("ACME blob has no certificate after the private key"); + } + let key = rustls::pki_types::PrivateKeyDer::Pkcs8(key.clone().into()); + TlsIdentity::from_chain(chain.to_vec(), key) + .context("building a serving configuration from the ACME certificate") +} + +/// Every PEM block's DER, in order. +/// +/// Delegated to a real PEM parser rather than split on `-----\n`. Only the +/// private key half of rustls-acme's blob is written by rustls-acme; the +/// certificate half is the ACME directory's HTTP response body verbatim +/// (`rustls_acme::state` concatenates the two), so its line endings are the +/// CA's choice. A hand-rolled split on LF drops every CRLF block *without +/// erroring*, which turned a valid issuance into an empty serving slot — and, +/// where a blob mixed the two, into a chain quietly missing its intermediate. +fn pem_blocks(pem: &[u8]) -> Result>> { + let blocks = pem::parse_many(pem).context("ACME blob is not valid PEM")?; + Ok(blocks + .into_iter() + .map(|block| block.into_contents()) + .collect()) +} + +/// Second PEM block of rustls-acme's cached blob. +pub fn leaf_from_pem(pem: &[u8]) -> Result> { + let mut blocks = pem_blocks(pem)?.into_iter(); + let _private_key = blocks + .next() + .context("ACME blob has no private key block")?; + blocks + .next() + .context("ACME blob has no certificate after the private key") +} + +/// AES-256-GCM with the object key as associated data. +/// +/// Binding the key means a sealed account blob cannot be moved over a +/// certificate blob, or one filesystem's cache swapped for another's, by +/// anyone who can write to the bucket. +fn seal(keys: &KeyMaterial, object_key: &str, plaintext: &[u8]) -> Result> { + use aws_lc_rs::aead::{Aad, Nonce, NONCE_LEN}; + use aws_lc_rs::rand::{SecureRandom, SystemRandom}; + + let mut nonce_bytes = [0u8; NONCE_LEN]; + SystemRandom::new() + .fill(&mut nonce_bytes) + .map_err(|_| anyhow::anyhow!("generating a nonce"))?; + + let mut buffer = plaintext.to_vec(); + keys.runtime_seal_key() + .seal_in_place_append_tag( + Nonce::assume_unique_for_key(nonce_bytes), + Aad::from(object_key.as_bytes()), + &mut buffer, + ) + .map_err(|_| anyhow::anyhow!("sealing the ACME cache entry"))?; + + let mut out = Vec::with_capacity(NONCE_LEN + buffer.len()); + out.extend_from_slice(&nonce_bytes); + out.extend_from_slice(&buffer); + Ok(out) +} + +fn unseal(keys: &KeyMaterial, object_key: &str, sealed: &[u8]) -> Result> { + use aws_lc_rs::aead::{Aad, Nonce, NONCE_LEN}; + + if sealed.len() <= NONCE_LEN { + anyhow::bail!("sealed ACME entry is too short to contain a nonce"); + } + let (nonce_bytes, body) = sealed.split_at(NONCE_LEN); + let mut nonce = [0u8; NONCE_LEN]; + nonce.copy_from_slice(nonce_bytes); + + let mut buffer = body.to_vec(); + let plaintext = keys + .runtime_seal_key() + .open_in_place( + Nonce::assume_unique_for_key(nonce), + Aad::from(object_key.as_bytes()), + &mut buffer, + ) + .map_err(|_| { + anyhow::anyhow!( + "the sealed ACME entry did not authenticate — it was written under a \ + different master key, or it has been tampered with" + ) + })?; + Ok(plaintext.to_vec()) +} + +#[async_trait::async_trait] +impl rustls_acme::CertCache for SealedAcmeCache { + type EC = anyhow::Error; + + async fn load_cert( + &self, + domains: &[String], + directory_url: &str, + ) -> Result>, Self::EC> { + let key = cache_key(&self.prefix, "cert", domains, directory_url); + let loaded = self.load(&key).await?; + if let Some(pem) = &loaded { + tracing::info!(?domains, "reusing the cached certificate"); + self.publish_leaf(pem); + } + Ok(loaded) + } + + async fn store_cert( + &self, + domains: &[String], + directory_url: &str, + cert: &[u8], + ) -> Result<(), Self::EC> { + let key = cache_key(&self.prefix, "cert", domains, directory_url); + tracing::info!(?domains, "storing a newly issued certificate"); + self.publish_leaf(cert); + self.store(&key, cert).await + } +} + +#[async_trait::async_trait] +impl rustls_acme::AccountCache for SealedAcmeCache { + type EA = anyhow::Error; + + async fn load_account( + &self, + contact: &[String], + directory_url: &str, + ) -> Result>, Self::EA> { + let key = cache_key(&self.prefix, "account", contact, directory_url); + self.load(&key).await + } + + async fn store_account( + &self, + contact: &[String], + directory_url: &str, + account: &[u8], + ) -> Result<(), Self::EA> { + let key = cache_key(&self.prefix, "account", contact, directory_url); + self.store(&key, account).await + } +} + +/// Everything needed to obtain and renew a certificate. +#[derive(Debug, Clone)] +pub struct AcmeConfig { + pub domains: Vec, + pub contacts: Vec, + /// ACME directory URL. `None` is Let's Encrypt production; point it at + /// staging or a local Pebble for testing, since production's rate limits + /// are low and per-domain. + pub directory: Option, + /// PEM trust root for the directory's *own* HTTPS certificate. + /// + /// `None` uses the public roots rustls-acme is built with, which is what + /// any real CA needs. A test CA like Pebble serves its API under a + /// certificate no public root signed, so its root has to be supplied — and + /// only a `testing` build can supply one, because a production enclave that + /// would take issuance orders from a CA of the operator's choosing is not + /// one this runtime offers. + pub directory_ca: Option>, + /// Key prefix for the sealed cache objects. + pub prefix: String, +} + +/// A running ACME client: the two TLS configurations it needs, and the slot it +/// publishes each certificate to. +pub struct Acme { + /// For TLS-ALPN-01 validation connections, which carry no HTTP at all — + /// the handshake *is* the proof, and the connection is then closed. + pub challenge_config: Arc, + /// Where each issued certificate is published, and where every connection + /// takes its serving identity. + /// + /// Deliberately not rustls-acme's own resolver: `process_cert` deploys a + /// renewed certificate to the resolver *before* the cache callback that + /// would publish it here, so serving from one while attesting the other + /// would attest a certificate the connection was never served. + pub certificate: CertificateSlot, + /// Fired once the listener is accepting, releasing the first order. + /// + /// A challenge is a connection *inbound* to :443, so ordering before the + /// socket exists guarantees the first attempt fails. Against Pebble that + /// costs a retry; against Let's Encrypt it spends one of five failed + /// validations per hour on every boot, and a crash-looping enclave would + /// lock itself out of issuance. + pub listening: Arc, +} + +impl std::fmt::Debug for Acme { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("Acme").finish_non_exhaustive() + } +} + +/// Start issuance and renewal in the background. +/// +/// Returns as soon as the machinery is running, *not* when a certificate +/// exists: issuance needs a round trip to the CA and an inbound challenge +/// connection, so blocking here would mean an enclave that cannot start until +/// the network the parent provides is already working. Until a certificate +/// arrives, TLS handshakes fail — which is honest, where serving a document +/// bound to a certificate nobody was served would not be. +pub fn start( + config: &AcmeConfig, + backend: Arc, + keys: Arc, +) -> Result { + anyhow::ensure!( + !config.domains.is_empty(), + "ACME needs at least one --acme-domain: a certificate is issued for names, \ + and the CA reaches those names to verify them" + ); + + let certificate = CertificateSlot::empty(); + let cache = SealedAcmeCache::new(backend, config.prefix.clone(), keys, certificate.clone()); + + let mut acme = rustls_acme::AcmeConfig::new(config.domains.clone()) + .contact(config.contacts.iter().map(|c| format!("mailto:{c}"))) + .cache(cache); + acme = match &config.directory { + Some(url) => acme.directory(url), + None => acme.directory_lets_encrypt(true), + }; + // Only when a root was supplied. Left alone, rustls-acme keeps the public + // roots it was compiled with, so the ordinary path gains no new trust and + // no new dependency. + if let Some(pem) = &config.directory_ca { + let mut roots = rustls::RootCertStore::empty(); + for der in pem_blocks(pem).context("reading the ACME directory's trust root")? { + roots + .add(rustls::pki_types::CertificateDer::from(der)) + .context("the ACME directory's trust root is not a usable certificate")?; + } + anyhow::ensure!( + !roots.is_empty(), + "the ACME directory trust root contains no certificates" + ); + tracing::warn!( + certificates = roots.len(), + "trusting a private root for the ACME directory; this is a test configuration" + ); + acme = acme.client_tls_config(Arc::new( + rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .context("selecting TLS protocol versions for the ACME client")? + .with_root_certificates(roots) + .with_no_client_auth(), + )); + } + + let state = acme.state(); + // The challenge config keeps no client auth: the CA validating + // TLS-ALPN-01 presents no certificate, and asking for one would be noise + // on the one handshake that must not fail. + let challenge_config = state.challenge_rustls_config(); + let mut state = state; + + let listening = Arc::new(tokio::sync::Notify::new()); + let wait_for_listener = listening.clone(); + + tracing::info!( + domains = ?config.domains, + directory = %config.directory.as_deref().unwrap_or("Let's Encrypt production"), + "ACME is configured; ordering starts once the listener is up" + ); + + tokio::spawn(async move { + use futures::StreamExt; + // Nothing is ordered until something can answer the challenge. The CA + // validates by connecting *in* on :443, so an order placed before the + // listener exists is an order that fails — reliably, on every boot. + // + // `Notify` and not a flag: it keeps a permit if the server binds first, + // so this cannot miss the signal by being slow to start. + wait_for_listener.notified().await; + tracing::info!("the listener is up; ordering a certificate"); + loop { + match state.next().await { + Some(Ok(ok)) => tracing::info!(event = ?ok, "ACME"), + // Logged at error and *not* fatal: a failed order is often a + // rate limit or a challenge that has not propagated, and + // rustls-acme retries with backoff. Killing the runtime would + // turn a transient CA problem into an outage. + Some(Err(e)) => tracing::error!(error = ?e, "ACME order failed; will retry"), + None => { + tracing::error!("the ACME state machine stopped; no renewals will happen"); + break; + } + } + } + }); + + Ok(Acme { + challenge_config, + certificate, + listening, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use s3fs_core::crypto::MasterSecret; + + fn keys() -> Arc { + Arc::new(KeyMaterial::derive(&MasterSecret::from_bytes([3u8; 32]), [1u8; 16]).unwrap()) + } + + #[test] + fn sealing_round_trips() { + let keys = keys(); + let sealed = seal(&keys, "acme/cert-abc", b"the private key and chain").unwrap(); + assert_ne!(sealed, b"the private key and chain"); + let opened = unseal(&keys, "acme/cert-abc", &sealed).unwrap(); + assert_eq!(opened, b"the private key and chain"); + } + + /// Binding the object key stops a sealed blob being moved from one slot to + /// another by anyone who can write to the bucket. + #[test] + fn a_blob_moved_to_another_key_does_not_open() { + let keys = keys(); + let sealed = seal(&keys, "acme/account-abc", b"account key").unwrap(); + assert!(unseal(&keys, "acme/cert-abc", &sealed).is_err()); + } + + #[test] + fn another_master_key_cannot_open_it() { + let sealed = seal(&keys(), "acme/cert-abc", b"secret").unwrap(); + let other = + Arc::new(KeyMaterial::derive(&MasterSecret::from_bytes([4u8; 32]), [1u8; 16]).unwrap()); + let err = unseal(&other, "acme/cert-abc", &sealed).unwrap_err(); + assert!(format!("{err:#}").contains("authenticate"), "{err:#}"); + } + + #[test] + fn a_tampered_blob_does_not_open() { + let keys = keys(); + let mut sealed = seal(&keys, "acme/cert-abc", b"secret").unwrap(); + let last = sealed.len() - 1; + sealed[last] ^= 0x01; + assert!(unseal(&keys, "acme/cert-abc", &sealed).is_err()); + } + + #[test] + fn a_truncated_blob_is_refused_rather_than_panicking() { + assert!(unseal(&keys(), "k", &[]).is_err()); + assert!(unseal(&keys(), "k", &[0u8; 8]).is_err()); + } + + /// Staging and production must not share a cache slot: a staging + /// certificate served in production is rejected by every client. + #[test] + fn the_directory_url_changes_the_cache_key() { + let domains = vec!["enclave.example".to_string()]; + let production = cache_key( + "", + "cert", + &domains, + "https://acme-v02.api.letsencrypt.org/directory", + ); + let staging = cache_key( + "", + "cert", + &domains, + "https://acme-staging-v02.api.letsencrypt.org/directory", + ); + assert_ne!(production, staging); + } + + #[test] + fn different_domains_get_different_keys() { + let a = cache_key("", "cert", &["a.example".to_string()], "d"); + let b = cache_key("", "cert", &["b.example".to_string()], "d"); + assert_ne!(a, b); + } + + #[test] + fn certificates_and_accounts_do_not_collide() { + let scope = vec!["enclave.example".to_string()]; + assert_ne!( + cache_key("", "cert", &scope, "d"), + cache_key("", "account", &scope, "d") + ); + } + + /// The leaf is the second PEM block. Reading the first would publish the + /// private key's bytes as the certificate hash — wrong, and alarming. + #[test] + fn the_leaf_is_taken_from_after_the_private_key() { + use base64::Engine; + let key_der = b"not really a key"; + let leaf_der = b"not really a certificate"; + let pem = format!( + "-----BEGIN PRIVATE KEY-----\n{}\n-----END PRIVATE KEY-----\n\ + -----BEGIN CERTIFICATE-----\n{}\n-----END CERTIFICATE-----\n", + base64::engine::general_purpose::STANDARD.encode(key_der), + base64::engine::general_purpose::STANDARD.encode(leaf_der), + ); + assert_eq!(leaf_from_pem(pem.as_bytes()).unwrap(), leaf_der); + } + + /// PEM is line-ending agnostic, and this blob is not all ours to format. + /// + /// rustls-acme concatenates its own LF private key with the ACME + /// directory's response body verbatim, so a CA that emits CRLF produces a + /// blob the runtime must still read. Splitting on `-----\n` dropped those + /// blocks silently: issuance succeeded, the serving slot stayed empty, and + /// the enclave answered no HTTPS at all. + #[test] + fn a_crlf_certificate_is_read_like_any_other() { + let key_der = vec![9u8; 8]; + let leaf_der = vec![4u8; 12]; + let block = |label: &str, der: &[u8], eol: &str| { + use base64::Engine; + let body = base64::engine::general_purpose::STANDARD.encode(der); + format!("-----BEGIN {label}-----{eol}{body}{eol}-----END {label}-----{eol}") + }; + + for eol in ["\n", "\r\n"] { + let blob = format!( + "{}{}", + block("PRIVATE KEY", &key_der, eol), + block("CERTIFICATE", &leaf_der, eol) + ); + assert_eq!( + pem_blocks(blob.as_bytes()).unwrap(), + vec![key_der.clone(), leaf_der.clone()], + "line ending {eol:?} must not change what is read" + ); + assert_eq!(leaf_from_pem(blob.as_bytes()).unwrap(), leaf_der); + } + } + + /// The quiet one: a chain that mixes endings must not lose a link. + /// + /// A dropped intermediate still parses and still serves, so nothing fails + /// until a client that needed it rejects the chain. + #[test] + fn a_chain_mixing_line_endings_keeps_every_link() { + use base64::Engine; + let der = |b: u8| vec![b; 10]; + let block = |label: &str, d: &[u8], eol: &str| { + let body = base64::engine::general_purpose::STANDARD.encode(d); + format!("-----BEGIN {label}-----{eol}{body}{eol}-----END {label}-----{eol}") + }; + let blob = format!( + "{}{}{}", + block("PRIVATE KEY", &der(1), "\n"), + block("CERTIFICATE", &der(2), "\n"), + block("CERTIFICATE", &der(3), "\r\n"), + ); + assert_eq!( + pem_blocks(blob.as_bytes()).unwrap(), + vec![der(1), der(2), der(3)], + "the CRLF intermediate was dropped" + ); + } + + #[test] + fn a_blob_without_a_certificate_is_an_error() { + let pem = "-----BEGIN PRIVATE KEY-----\nAAAA\n-----END PRIVATE KEY-----\n"; + assert!(leaf_from_pem(pem.as_bytes()).is_err()); + } + + #[test] + fn a_slot_reports_what_was_last_set() { + let first = Arc::new(TlsIdentity::self_signed(&["first.test".into()]).expect("first")); + let renewed = + Arc::new(TlsIdentity::self_signed(&["renewed.test".into()]).expect("renewed")); + + let slot = CertificateSlot::empty(); + assert!(slot.get().is_none()); + assert!(slot.leaf().is_none()); + + slot.set(first.clone()); + assert_eq!(slot.leaf(), Some(first.certificate_der.clone())); + + slot.set(renewed.clone()); + assert_eq!(slot.leaf(), Some(renewed.certificate_der.clone())); + } + + /// The property the whole slot exists for: an identity taken out of it is a + /// configuration *and* the leaf that configuration presents, so a + /// connection served from one can be attested with the other. Reading them + /// separately is what let a renewal attest a certificate the connection was + /// never served. + #[test] + fn an_identity_pairs_a_config_with_the_leaf_it_presents() { + let identity = Arc::new(TlsIdentity::self_signed(&["paired.test".into()]).expect("built")); + let slot = CertificateSlot::fixed(identity.clone()); + + let loaded = slot.get().expect("an identity"); + assert_eq!(loaded.certificate_der, identity.certificate_der); + assert!(Arc::ptr_eq(&loaded.config, &identity.config)); + + // A renewal replaces the pair; a connection holding the old one is + // unaffected, which is the case that matters. + let renewed = Arc::new(TlsIdentity::self_signed(&["renewed.test".into()]).expect("built")); + slot.set(renewed); + assert_eq!( + loaded.certificate_der, identity.certificate_der, + "a loaded identity must not change under a renewal" + ); + } +} diff --git a/runtime/src/serve/attest.rs b/runtime/src/serve/attest.rs new file mode 100644 index 0000000..fce1f12 --- /dev/null +++ b/runtime/src/serve/attest.rs @@ -0,0 +1,624 @@ +//! A proof, on every `/auth/*` response, that this connection is terminated by +//! this enclave. +//! +//! ```text +//! client nonce ──┐ +//! ├──▶ NSM ──▶ COSE_Sign1 ──▶ x-enclave-attestation +//! this connection's leaf ──┘ +//! ``` +//! +//! # What it proves +//! +//! That the enclave holding the private key for **the certificate this +//! connection was served** is alive now, and saw a nonce the client chose. A +//! client checks it by hashing the certificate from its *own* handshake and +//! comparing. Putting that on the auth exchange itself, rather than behind a +//! separate attestation route, is what makes it need no second round trip and +//! stops it drifting from the connection it describes. +//! +//! # Which responses carry one +//! +//! Every response to a path under `/auth/`, whatever its method or status — the +//! challenge, the token, a refusal, and a 405 for `GET /auth/`, which is how a +//! client attests the enclave without asking it for anything (`nitro-attest` +//! does exactly that). The document is made before routing, so it cannot +//! depend on what the route answers. +//! +//! Nothing else carries one. A guest response is served on a connection whose +//! certificate the client already verified on the auth exchange and pinned, and +//! TLS proves the peer still holds that certificate's key — so a second +//! document would only spend an NSM signature to repeat the first. The header +//! is stripped from those responses, so a guest cannot supply one of its own. +//! +//! A request without a valid `x-enclave-nonce` is refused with a 400 before +//! routing, and carries no document either: there is no client nonce to bind. +//! +//! # What it does not prove +//! +//! **Nothing about the response body.** The document is generated before the +//! guest runs and binds only the nonce and the certificate. It is not a receipt +//! for what the guest said, and adding one would mean a new `user_data` layout +//! and a verifier that understands it. +//! +//! **Nothing before the request was sent.** By the time a client sees the +//! proof, its request is already inside the enclave. A client that must know +//! first sends a nonced request, verifies, and reuses **the same +//! connection** — which proves more than a separate round trip could, since the +//! binding is per-connection. +//! +//! The client also has to do its part, and the runtime cannot make it: generate +//! the nonce from a CSPRNG per request, compare the document's nonce to the one +//! sent, and treat a *missing* header as a failure. Without the last of those, +//! an attacker who strips the header downgrades every client that does not +//! check. + +use std::sync::Arc; + +use hyper::header::{HeaderName, HeaderValue}; +use hyper::HeaderMap; +use nitro_attestation::AttestationHashes; +use nitro_nsm::{AttestationRequest, Nsm}; +use tokio::sync::Semaphore; + +/// The client's nonce, base64url without padding — the encoding the auth +/// headers already use. +pub const NONCE_HEADER: HeaderName = HeaderName::from_static("x-enclave-nonce"); + +/// The document, base64 standard — one line, one value, decodable by anything +/// that can read a header. +pub const ATTESTATION_HEADER: HeaderName = HeaderName::from_static("x-enclave-attestation"); + +/// A nonce shorter than this is not worth calling a nonce. +pub const MIN_NONCE_BYTES: usize = 8; +/// The NSM's own request field limit is far higher; this is a sanity bound. +pub const MAX_NONCE_BYTES: usize = 64; + +/// Ceiling on the encoded header. +/// +/// A real Nitro document is roughly 4–6 KiB of CBOR — most of it the CA bundle, +/// which is inside the signed payload and cannot be trimmed — so base64 lands +/// near 6–8 KiB. Checked once at startup rather than per request, because a +/// document too large for a header is a deployment fact, not a request-time +/// surprise. +pub const MAX_DOCUMENT_HEADER_BYTES: usize = 16 * 1024; + +/// Documents in flight at once. +/// +/// Each document is an ECDSA P-384 signature on the device, and the device is +/// one device. This is what stops attestation exhausting the NSM or the +/// blocking pool — and, because it is now on the path of *every* request, it is +/// also the runtime's throughput ceiling. +const CONCURRENT_DOCUMENTS: usize = 4; + +/// Why a nonce was not acceptable. +/// +/// Every variant is a client mistake and says so plainly — unlike the auth +/// gate, there is nothing here an attacker learns from the distinction. +#[derive(Debug, PartialEq, Eq)] +pub enum NonceError { + Missing, + NotAscii, + NotBase64Url, + TooShort(usize), + TooLong(usize), +} + +impl std::fmt::Display for NonceError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + NonceError::Missing => write!( + f, + "every request must carry {NONCE_HEADER}: \ + {MIN_NONCE_BYTES}..={MAX_NONCE_BYTES} random bytes, base64url, unpadded" + ), + NonceError::NotAscii => write!(f, "{NONCE_HEADER} is not printable ASCII"), + NonceError::NotBase64Url => { + write!(f, "{NONCE_HEADER} is not unpadded base64url") + } + NonceError::TooShort(n) => write!( + f, + "{NONCE_HEADER} decoded to {n} bytes; the minimum is {MIN_NONCE_BYTES}" + ), + NonceError::TooLong(n) => write!( + f, + "{NONCE_HEADER} decoded to {n} bytes; the maximum is {MAX_NONCE_BYTES}" + ), + } + } +} + +/// A nonce the client chose, already checked. +/// +/// A newtype so a validated nonce cannot be confused with arbitrary bytes on +/// the way to the device. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ClientNonce(Vec); + +impl ClientNonce { + pub fn as_bytes(&self) -> &[u8] { + &self.0 + } +} + +/// Read and check the nonce. +/// +/// Uses the first value: `HeaderMap` keeps every copy a client sent, and +/// `get` returns the first, deterministically. A client that sends two knows +/// which one was bound — it is the one it sent first — rather than having to +/// guess at the runtime's choice. +pub fn nonce_from_headers(headers: &HeaderMap) -> Result { + use base64::Engine as _; + + let value = headers.get(&NONCE_HEADER).ok_or(NonceError::Missing)?; + let text = value.to_str().map_err(|_| NonceError::NotAscii)?; + let bytes = base64::engine::general_purpose::URL_SAFE_NO_PAD + .decode(text.trim()) + .map_err(|_| NonceError::NotBase64Url)?; + + if bytes.len() < MIN_NONCE_BYTES { + return Err(NonceError::TooShort(bytes.len())); + } + if bytes.len() > MAX_NONCE_BYTES { + return Err(NonceError::TooLong(bytes.len())); + } + Ok(ClientNonce(bytes)) +} + +/// Why a document could not be produced. +#[derive(Debug)] +pub enum AttestError { + /// No certificate on this connection — plaintext, or nothing issued yet. + NoCertificate, + /// The device refused, or the runtime is shutting down. + Device(String), + /// The document does not fit in a header. + TooLarge(usize), +} + +impl std::fmt::Display for AttestError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + AttestError::NoCertificate => { + write!(f, "this connection has no certificate to bind") + } + AttestError::Device(e) => write!(f, "the security module did not answer: {e}"), + AttestError::TooLarge(n) => write!( + f, + "the document is {n} bytes encoded, over the {MAX_DOCUMENT_HEADER_BYTES}-byte \ + header ceiling" + ), + } + } +} + +/// Produces one document per response. +pub struct ResponseAttestor { + nsm: Arc, + guest: [u8; 32], + limit: Arc, +} + +impl std::fmt::Debug for ResponseAttestor { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("ResponseAttestor") + .field("guest_sha256", &hex::encode(self.guest)) + .field("available", &self.limit.available_permits()) + .finish_non_exhaustive() + } +} + +impl ResponseAttestor { + pub fn new(nsm: Arc, guest_component: &[u8]) -> Self { + ResponseAttestor { + nsm, + guest: nitro_attestation::sha256(guest_component), + limit: Arc::new(Semaphore::new(CONCURRENT_DOCUMENTS)), + } + } + + /// A document binding `nonce` and the leaf this connection was served. + /// + /// The permit is taken **before** the blocking call, not inside it: tokio's + /// blocking pool queues without bound, so acquiring inside would convert a + /// burst into thread growth rather than backpressure. + /// + /// It is then *moved into* the closure, and this is load-bearing. A + /// `spawn_blocking` task cannot be cancelled — dropping its `JoinHandle` + /// only detaches it, and the ioctl runs to completion regardless. A permit + /// merely borrowed by this future would therefore be released the moment a + /// caller went away while the device was still working, letting the next + /// request start a call the limit was supposed to hold back. Since this + /// runs before the request is routed, and so before any auth gate, a peer + /// that connects and disconnects in a loop could drive the device past + /// `CONCURRENT_DOCUMENTS` without ever authenticating. Owning the permit + /// ties its lifetime to the work rather than to the waiter. + pub async fn document( + &self, + certificate: Option<&[u8]>, + nonce: &ClientNonce, + ) -> Result { + let leaf = certificate.ok_or(AttestError::NoCertificate)?; + // Constructed field-wise, not via `AttestationHashes::new`: `self.guest` + // is already the digest of the component, and `new` hashes what it is + // given. Passing it there would bind sha256(sha256(guest)) — which + // verifies against nothing a client can compute. + let user_data = AttestationHashes { + tls_certificate: nitro_attestation::sha256(leaf), + guest: self.guest, + } + .serialize(); + let request = AttestationRequest::with_user_data(user_data).nonce(nonce.0.clone()); + + let permit = self + .limit + .clone() + .acquire_owned() + .await + .map_err(|_| AttestError::Device("attestation is shutting down".into()))?; + + // `Nsm::attest` is a synchronous ioctl. Left inline it would hold a + // tokio worker thread for the whole signature, which on a per-response + // path is every worker. + let nsm = self.nsm.clone(); + let document = tokio::task::spawn_blocking(move || { + // Held until the ioctl returns, not until this caller does. + let _permit = permit; + nsm.attest(&request) + }) + .await + .map_err(|e| AttestError::Device(format!("attestation task: {e}")))? + .map_err(|e| AttestError::Device(format!("{e:#}")))?; + + encode(&document) + } + + /// Check at startup that a document from this device fits in a header. + /// + /// Boot is the only place this can be a hard failure rather than a + /// per-request surprise, and a runtime that cannot attest its responses + /// should not accept requests it will have to refuse. It is also the first + /// call to the device, so an NSM that will not answer at all is found here + /// rather than on a client's first request. + /// + /// Deliberately does **not** take the serving certificate. On the ACME path + /// there is none yet — issuance needs a round trip to the CA and an inbound + /// challenge, both of which happen after this — and refusing to start then + /// would mean no ACME deployment could ever boot. A placeholder is sound + /// because only the leaf's 32-byte SHA-256 reaches `user_data`, so the + /// document is the same size whatever certificate is eventually served. + pub async fn verify_fits(&self) -> Result { + let probe = ClientNonce(vec![0u8; MAX_NONCE_BYTES]); + let header = self.document(Some(b"boot-time size probe"), &probe).await?; + Ok(header.len()) + } +} + +/// Base64 the document into a header value. +fn encode(document: &[u8]) -> Result { + use base64::Engine as _; + let encoded = base64::engine::general_purpose::STANDARD.encode(document); + if encoded.len() > MAX_DOCUMENT_HEADER_BYTES { + return Err(AttestError::TooLarge(encoded.len())); + } + // Base64 is header-safe by construction, so this cannot fail — but a + // device returning something unexpected should not panic a request path. + HeaderValue::from_str(&encoded) + .map_err(|_| AttestError::Device("document is not a header value".into())) +} + +/// Put the document on a response, and take ownership of the name. +/// +/// `insert`, never `append`: a guest can set any response header it likes, and +/// `HeaderMap::insert` replaces *every* existing value for the name, so a guest +/// that sets three copies ends up with none of its own. This is the response +/// side of the discipline `apply_tenant` follows for requests. +/// +/// `cache-control: no-store` goes with it. A cached attested response is a +/// replayed document, bound to a nonce the next reader never chose. +pub fn attach(response: &mut hyper::Response, document: HeaderValue) { + let headers = response.headers_mut(); + headers.insert(ATTESTATION_HEADER, document); + headers.insert( + hyper::header::CACHE_CONTROL, + HeaderValue::from_static("no-store"), + ); +} + +/// Take the header away from a response the runtime did not attest. +/// +/// The header is runtime-owned, and it was only ever guest-proof because +/// [`attach`] overwrote it on every response. Once some responses are not +/// attested — an operation whose caller already identified the enclave on the +/// `/auth/` exchange and pinned its certificate — "overwritten" stops being +/// true for them, and a guest that sets it would be handing the client a +/// document of its own choosing under the runtime's name. +/// +/// So the header is removed rather than left alone. A guest cannot speak here +/// whether or not the runtime is speaking here, which is the property; that it +/// used to hold as a side effect of always attesting was luck, not design. +pub fn strip(response: &mut hyper::Response) { + response.headers_mut().remove(ATTESTATION_HEADER); +} + +#[cfg(test)] +mod tests { + use super::*; + use base64::Engine as _; + + fn headers_with(value: &str) -> HeaderMap { + let mut headers = HeaderMap::new(); + headers.insert(NONCE_HEADER, HeaderValue::from_str(value).unwrap()); + headers + } + + fn encoded(bytes: &[u8]) -> String { + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(bytes) + } + + /// A cancelled caller must not hand its permit to the next request while + /// the device is still working on its behalf. + /// + /// `spawn_blocking` cannot be cancelled, so the ioctl outlives the future + /// that asked for it. If the permit were merely borrowed by that future it + /// would be released on cancellation, and `CONCURRENT_DOCUMENTS` would cap + /// nothing: with a limit of four this drove eight concurrent calls. The + /// permit is owned by the closure precisely so this cannot happen — and + /// since attestation runs ahead of the auth gate, an unauthenticated peer + /// is the one who would otherwise drive it. + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] + async fn a_cancelled_caller_does_not_release_the_device() { + use std::sync::atomic::{AtomicUsize, Ordering}; + + /// Reports how many `attest` calls are inside the device at once. + #[derive(Debug)] + struct SlowNsm { + inflight: Arc, + peak: Arc, + } + + impl nitro_nsm::Nsm for SlowNsm { + fn get_random(&self, _buf: &mut [u8]) -> anyhow::Result<()> { + unreachable!("attestation does not draw entropy") + } + + fn attest(&self, _request: &AttestationRequest) -> anyhow::Result> { + let now = self.inflight.fetch_add(1, Ordering::SeqCst) + 1; + self.peak.fetch_max(now, Ordering::SeqCst); + // Long enough that a cancelled caller is gone well before the + // "device" returns, which is the whole point of the test. + std::thread::sleep(std::time::Duration::from_millis(300)); + self.inflight.fetch_sub(1, Ordering::SeqCst); + Ok(vec![0u8; 16]) + } + + fn describe_pcr(&self, _index: u16) -> anyhow::Result { + unreachable!("attestation does not read PCRs") + } + + fn extend_pcr(&self, _index: u16, _data: &[u8]) -> anyhow::Result> { + unreachable!("attestation does not extend PCRs") + } + + fn lock_pcr(&self, _index: u16) -> anyhow::Result<()> { + unreachable!("attestation does not lock PCRs") + } + + fn describe(&self) -> String { + "a deliberately slow test device".into() + } + } + + let inflight = Arc::new(AtomicUsize::new(0)); + let peak = Arc::new(AtomicUsize::new(0)); + let attestor = Arc::new(ResponseAttestor::new( + Arc::new(SlowNsm { + inflight: inflight.clone(), + peak: peak.clone(), + }), + b"guest", + )); + + let call = |a: Arc| async move { + let nonce = ClientNonce(vec![0u8; MIN_NONCE_BYTES]); + a.document(Some(b"leaf"), &nonce).await + }; + + // Take every permit, then abandon each caller mid-ioctl. + let abandoned: Vec<_> = (0..CONCURRENT_DOCUMENTS) + .map(|_| tokio::spawn(call(attestor.clone()))) + .collect(); + tokio::time::sleep(std::time::Duration::from_millis(80)).await; + for handle in &abandoned { + handle.abort(); + } + tokio::time::sleep(std::time::Duration::from_millis(40)).await; + + assert_eq!( + attestor.limit.available_permits(), + 0, + "cancelling the callers released permits the device is still using" + ); + + // Fresh callers must queue behind the work that is still running. + let queued: Vec<_> = (0..CONCURRENT_DOCUMENTS) + .map(|_| tokio::spawn(call(attestor.clone()))) + .collect(); + tokio::time::sleep(std::time::Duration::from_millis(150)).await; + let observed = peak.load(Ordering::SeqCst); + for handle in queued { + let _ = handle.await; + } + + assert!( + observed <= CONCURRENT_DOCUMENTS, + "{observed} concurrent attestations against a limit of {CONCURRENT_DOCUMENTS}" + ); + } + + #[test] + fn a_well_formed_nonce_is_accepted() { + let nonce = [7u8; 20]; + let parsed = nonce_from_headers(&headers_with(&encoded(&nonce))).expect("valid"); + assert_eq!(parsed.as_bytes(), &nonce); + } + + /// A request with no nonce reaches nothing: there is no value to bind, and + /// one the runtime chose would prove nothing to the client. + #[test] + fn a_missing_nonce_is_refused() { + assert_eq!( + nonce_from_headers(&HeaderMap::new()), + Err(NonceError::Missing) + ); + } + + #[test] + fn the_boundaries_are_where_they_are_documented() { + let ok = |n: usize| nonce_from_headers(&headers_with(&encoded(&vec![0u8; n]))); + assert!(ok(MIN_NONCE_BYTES).is_ok(), "the minimum must be allowed"); + assert!(ok(MAX_NONCE_BYTES).is_ok(), "the maximum must be allowed"); + assert_eq!( + ok(MIN_NONCE_BYTES - 1), + Err(NonceError::TooShort(MIN_NONCE_BYTES - 1)) + ); + assert_eq!( + ok(MAX_NONCE_BYTES + 1), + Err(NonceError::TooLong(MAX_NONCE_BYTES + 1)) + ); + } + + /// Standard base64 is refused rather than quietly accepted: one encoding, + /// so a client cannot be right by accident and wrong later. + #[test] + fn only_unpadded_base64url_is_accepted() { + let nonce = [0xffu8; 16]; + let standard = base64::engine::general_purpose::STANDARD.encode(nonce); + assert!(standard.contains('/') || standard.contains('+') || standard.ends_with('=')); + assert_eq!( + nonce_from_headers(&headers_with(&standard)), + Err(NonceError::NotBase64Url) + ); + assert_eq!( + nonce_from_headers(&headers_with("not base64 at all!")), + Err(NonceError::NotBase64Url) + ); + } + + /// Whatever the guest set, the runtime's value is the only one that + /// survives — including when the guest sent several. The response-side + /// mirror of `a_client_cannot_smuggle_a_tenant_header`. + #[test] + fn a_guest_cannot_own_the_attestation_header() { + let mut response = hyper::Response::new(()); + for forged in ["forged-one", "forged-two", "forged-three"] { + response + .headers_mut() + .append(ATTESTATION_HEADER, HeaderValue::from_static(forged)); + } + response.headers_mut().insert( + hyper::header::CACHE_CONTROL, + HeaderValue::from_static("public, max-age=3600"), + ); + + attach(&mut response, HeaderValue::from_static("the-real-document")); + + let values: Vec<_> = response + .headers() + .get_all(ATTESTATION_HEADER) + .iter() + .collect(); + assert_eq!(values.len(), 1, "the guest's copies survived: {values:?}"); + assert_eq!(values[0], "the-real-document"); + + // A cached attested response is a replayed document. + assert_eq!( + response + .headers() + .get(hyper::header::CACHE_CONTROL) + .unwrap(), + "no-store", + "the guest's caching directive survived" + ); + } + + #[tokio::test] + async fn a_document_binds_the_nonce_and_this_connection_certificate() { + let nsm = Arc::new(nitro_nsm::fake::FakeNsm::new()); + let attestor = ResponseAttestor::new(nsm.clone(), b"a guest"); + let nonce = nonce_from_headers(&headers_with(&encoded(&[3u8; 24]))).unwrap(); + + let header = attestor + .document(Some(b"this connection's leaf"), &nonce) + .await + .expect("a document"); + assert!(!header.is_empty()); + + let request = nsm + .last_attestation_request + .lock() + .unwrap() + .clone() + .expect("the device was asked"); + assert_eq!(request.nonce.as_deref(), Some(nonce.as_bytes())); + assert_eq!( + request.user_data.as_deref(), + // Computed the way a *client* computes it — from the certificate it + // was served and the component bytes it knows — not from the + // runtime's own intermediate values. The earlier version of this + // assertion mirrored a double-hash bug in the code above and so + // agreed with it; an integration test against a real verifier + // caught what this one could not. + Some(&AttestationHashes::new(b"this connection's leaf", b"a guest").serialize()[..]), + "the document must bind the certificate this connection was served" + ); + } + + /// Plaintext, or a certificate that has not been issued yet. Refusing is + /// the only honest answer: there is nothing to bind. + #[tokio::test] + async fn a_connection_without_a_certificate_cannot_be_attested() { + let attestor = ResponseAttestor::new(Arc::new(nitro_nsm::fake::FakeNsm::new()), b"g"); + let nonce = nonce_from_headers(&headers_with(&encoded(&[1u8; 16]))).unwrap(); + assert!(matches!( + attestor.document(None, &nonce).await, + Err(AttestError::NoCertificate) + )); + } + + /// A device that will not answer fails the request rather than releasing + /// an unattested response as though it were attested. + #[tokio::test] + async fn a_refusing_device_is_an_error_not_a_missing_header() { + let nsm = Arc::new(nitro_nsm::fake::FakeNsm::new()); + nsm.empty.store(true, std::sync::atomic::Ordering::SeqCst); + let attestor = ResponseAttestor::new(nsm, b"g"); + let nonce = nonce_from_headers(&headers_with(&encoded(&[1u8; 16]))).unwrap(); + assert!(matches!( + attestor.document(Some(b"leaf"), &nonce).await, + Err(AttestError::Device(_)) + )); + } + + /// The boot check exists so an oversized document stops the runtime rather + /// than every request. + #[tokio::test] + async fn an_oversized_document_is_refused() { + let nsm = Arc::new(nitro_nsm::fake::FakeNsm::new()); + *nsm.attestation.lock().unwrap() = vec![0u8; MAX_DOCUMENT_HEADER_BYTES]; + let attestor = ResponseAttestor::new(nsm, b"g"); + assert!(matches!( + attestor.verify_fits().await, + Err(AttestError::TooLarge(_)) + )); + } + + /// The check must not need the serving certificate. An ACME deployment has + /// none at boot — issuance happens later — so a check that asked for one + /// would refuse to start exactly the deployments this runtime is for. + #[tokio::test] + async fn the_boot_check_does_not_need_a_certificate_yet() { + let nsm = Arc::new(nitro_nsm::fake::FakeNsm::new()); + let attestor = ResponseAttestor::new(nsm, b"g"); + assert!(attestor.verify_fits().await.is_ok()); + } +} diff --git a/runtime/src/serve/client.rs b/runtime/src/serve/client.rs new file mode 100644 index 0000000..d137a50 --- /dev/null +++ b/runtime/src/serve/client.rs @@ -0,0 +1,86 @@ +//! Which tenant a request speaks for, as the guest is told. +//! +//! The runtime resolves a tenant from a verified WebAuthn assertion — see +//! [`crate::auth`] — and injects it here. A guest can read this header and can +//! never write one: it is overwritten on every request before the guest sees +//! anything, so what arrives is the runtime's word rather than the client's. +//! +//! There used to be a second answer to "who is calling": a TLS client +//! certificate, hashed to its SPKI. It was removed when passkeys arrived, and +//! not because it was broken. A certificate proves possession of a key for the +//! life of a connection; it cannot say that a person approved *this* +//! transaction, and two independent ways to become a tenant is where an +//! authorization bug grows. + +use hyper::header::{HeaderName, HeaderValue}; + +/// The tenant the runtime resolved. Written by the runtime, never read from a +/// client. +pub const X_ENCLAVE_TENANT: HeaderName = HeaderName::from_static("x-enclave-tenant"); + +/// Set or remove the tenant header, whatever the client sent. +/// +/// `insert`, not `append`: a client that sent the header three times would +/// otherwise leave two of its own behind for a guest reading `[0]`. And the +/// `None` arm removes rather than skips, so an unauthenticated request cannot +/// carry a tenant into the guest by claiming one. +/// +/// This cannot live in `EgressPolicy::is_forbidden_header`, which runs *inside* +/// `new_incoming_request` after injection — it could only delete the header, +/// silently, with no error anywhere. +pub fn apply_tenant(tenant: Option<&[u8; 16]>, req: &mut hyper::Request) { + match tenant { + Some(id) => { + let value = HeaderValue::from_str(&hex::encode(id)) + .expect("hex is always a valid header value"); + req.headers_mut().insert(X_ENCLAVE_TENANT, value); + } + None => { + req.headers_mut().remove(X_ENCLAVE_TENANT); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn request_with(values: &[&str]) -> hyper::Request<()> { + let mut builder = hyper::Request::builder(); + for v in values { + builder = builder.header(X_ENCLAVE_TENANT, *v); + } + builder.body(()).expect("well-formed request") + } + + /// The forgery. A client sending the header itself must not be believed — + /// and sending it three times must not leave one behind, because a guest + /// reading the first value would get the attacker's. + #[test] + fn a_client_cannot_smuggle_a_tenant_header() { + let mut req = request_with(&["deadbeef", "deadbeef", "deadbeef"]); + apply_tenant(Some(&[0xab; 16]), &mut req); + let seen: Vec<_> = req.headers().get_all(X_ENCLAVE_TENANT).iter().collect(); + assert_eq!(seen.len(), 1, "a client-supplied header survived"); + assert_eq!(seen[0], &hex::encode([0xab; 16])); + } + + /// The shorter route to the same forgery: no tenant at all, so the + /// client's own header must be removed rather than passed through. + #[test] + fn an_unauthenticated_request_carries_no_tenant() { + let mut req = request_with(&["deadbeef"]); + apply_tenant(None, &mut req); + assert!(req.headers().get(X_ENCLAVE_TENANT).is_none()); + } + + #[test] + fn a_tenant_reaches_the_guest_as_hex() { + let mut req = request_with(&[]); + apply_tenant(Some(&[0x01; 16]), &mut req); + assert_eq!( + req.headers().get(X_ENCLAVE_TENANT).unwrap(), + &"01".repeat(16) + ); + } +} diff --git a/runtime/src/serve/egress.rs b/runtime/src/serve/egress.rs new file mode 100644 index 0000000..f1971ab --- /dev/null +++ b/runtime/src/serve/egress.rs @@ -0,0 +1,473 @@ +//! The origins a guest may reach, when a deployment names any. +//! +//! A guest reaches nothing by default — see [`super::EgressPolicy`]. Some guests +//! cannot do their job from inside that: a wallet cosigner that holds a pre-signed +//! renewal has to talk to the service it renews with, and the only alternative is +//! routing the conversation through a phone that may be off. So a deployment may +//! name **origins** — scheme, host and port, nothing else — and a guest may send +//! `wasi:http` requests to exactly those. +//! +//! What that does and does not open: +//! +//! - **Origins, compared exactly.** `https://asp.example.com` admits that host on +//! 443 and nothing else: not a subdomain, not another port, not plaintext to +//! the same name. No wildcards, no paths, no user info. +//! - **HTTPS is verified against the public web PKI** (webpki roots, the same +//! store the runtime's own FCM client uses). A plaintext `http://` origin is +//! admitted only when named as one — which is for a local development stack +//! and is measured like every other choice. +//! - **It is part of the image.** The list reaches the runtime as image +//! environment, so PCR0 covers it: a client verifying an enclave learns where +//! its guest can send traffic, and changing it is a new image. +//! - **It is a channel out of the enclave.** Whatever the guest puts in a request +//! leaves the attested boundary. Name only services the guest has to reach, and +//! treat the guest code as the thing deciding what they are sent. +//! +//! Requests travel over HTTP/1.1, from a fresh connection per request. + +use std::collections::BTreeSet; +use std::fmt; +use std::sync::Arc; + +use anyhow::{bail, Context, Result}; +use http_body_util::BodyExt; +use tokio::net::TcpStream; +use tokio::time::timeout; +use wasmtime_wasi_http::io::TokioIo; +use wasmtime_wasi_http::p2::{ + bindings::http::types::ErrorCode, + body::{HyperIncomingBody, HyperOutgoingBody}, + hyper_request_error, + types::{HostFutureIncomingResponse, IncomingResponse, OutgoingRequestConfig}, +}; + +/// One origin a guest may reach. +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord)] +pub struct Origin { + pub tls: bool, + /// Lowercase. + pub host: String, + pub port: u16, +} + +impl Origin { + /// Parse `https://host[:port]` or `http://host[:port]`. Anything more — a + /// path, a query, user info, a wildcard — is refused rather than ignored, + /// because an entry that reads as narrower than it is would be the worst kind + /// of mistake in this list. + pub fn parse(s: &str) -> Result { + let s = s.trim(); + let (tls, rest) = if let Some(rest) = s.strip_prefix("https://") { + (true, rest) + } else if let Some(rest) = s.strip_prefix("http://") { + (false, rest) + } else { + bail!("guest egress origin {s:?} must start with https:// or http://"); + }; + let rest = rest.strip_suffix('/').unwrap_or(rest); + if rest.is_empty() || rest.contains(['/', '?', '#', '@', '*']) { + bail!("guest egress origin {s:?} must be scheme://host[:port] and nothing else"); + } + let (host, port) = match rest.rsplit_once(':') { + Some((host, port)) if !host.contains(']') || host.ends_with(']') => ( + host, + port.parse::() + .ok() + .filter(|p| *p != 0) + .with_context(|| format!("guest egress origin {s:?} has an invalid port"))?, + ), + _ => (rest, if tls { 443 } else { 80 }), + }; + if host.is_empty() + || !host + .chars() + .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '-' | '[' | ']' | ':')) + { + bail!("guest egress origin {s:?} has an invalid host"); + } + Ok(Origin { + tls, + host: host.to_ascii_lowercase(), + port, + }) + } + + /// The origin a request is addressed to, or `None` if it names none. + fn of(request: &hyper::Request, use_tls: bool) -> Option { + let uri = request.uri(); + let tls = match uri.scheme_str() { + Some("https") => true, + Some("http") => false, + None => use_tls, + Some(_) => return None, + }; + let authority = uri.authority()?; + Some(Origin { + tls, + host: authority.host().to_ascii_lowercase(), + port: authority.port_u16().unwrap_or(if tls { 443 } else { 80 }), + }) + } +} + +impl fmt::Display for Origin { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + let scheme = if self.tls { "https" } else { "http" }; + write!(f, "{scheme}://{}:{}", self.host, self.port) + } +} + +/// The origins a deployment admits. +#[derive(Debug, Default)] +pub struct EgressAllowlist { + origins: BTreeSet, + tls: Option>, +} + +impl EgressAllowlist { + /// Parse every entry, failing on the first that does not parse. Empty + /// entries are dropped: an image built with no origins still sets the + /// variable, to nothing. + pub fn parse>(entries: &[S]) -> Result { + let origins = entries + .iter() + .map(|e| e.as_ref().trim()) + .filter(|e| !e.is_empty()) + .map(Origin::parse) + .collect::>>()?; + let tls = if origins.iter().any(|o| o.tls) { + Some(Arc::new(crate::notify::web_pki_client_config()?)) + } else { + None + }; + Ok(EgressAllowlist { origins, tls }) + } + + pub fn is_empty(&self) -> bool { + self.origins.is_empty() + } + + pub fn origins(&self) -> impl Iterator { + self.origins.iter() + } + + /// Whether `request` is addressed to an admitted origin. + /// + /// The scheme the URI names and the TLS flag the guest set must agree: a + /// guest cannot name `https://` and have it sent in the clear, or the reverse. + pub fn admits( + &self, + request: &hyper::Request, + config: &OutgoingRequestConfig, + ) -> Option { + let origin = Origin::of(request, config.use_tls)?; + (origin.tls == config.use_tls && self.origins.contains(&origin)).then_some(origin) + } + + /// Send from the RUNTIME's own code rather than on a guest's behalf. + /// + /// Same connection, same TLS, same allowlist — the caller has already had + /// `origin` admitted. What differs is only the shape of the answer: a guest + /// gets a `wasi:http` future it will poll, and the runtime gets a response + /// it can await. Used by [`crate::stream`], which holds connections that no + /// guest could hold for itself. + pub async fn send_direct( + &self, + origin: Origin, + request: hyper::Request, + first_byte_timeout: std::time::Duration, + ) -> std::result::Result { + let config = OutgoingRequestConfig { + use_tls: origin.tls, + connect_timeout: std::time::Duration::from_secs(10), + first_byte_timeout, + // A held stream is quiet between events; a service that has said + // nothing for this long is one worth reconnecting to. + between_bytes_timeout: std::time::Duration::from_secs(300), + }; + let incoming = send(origin, self.tls.clone(), request, config).await?; + Ok(HeldResponse { + response: incoming.resp, + _worker: incoming.worker, + }) + } + + /// Send `request` to `origin`, which [`Self::admits`] has already approved. + pub fn send( + &self, + origin: Origin, + request: hyper::Request, + config: OutgoingRequestConfig, + ) -> HostFutureIncomingResponse { + let tls = self.tls.clone(); + let handle = + wasmtime_wasi::runtime::spawn( + async move { Ok(send(origin, tls, request, config).await) }, + ); + HostFutureIncomingResponse::pending(handle) + } +} + +/// A response, with the connection that is still feeding it. +/// +/// The body of a `hyper` response is fed by a task driving the connection, and that task is +/// abort-on-drop. Handing back the response alone therefore ends the body the moment the call +/// returns — for a request/response exchange nobody notices, because the whole body is already +/// buffered, but for a stream held open it means the stream ends immediately and cleanly. That is +/// not a hypothetical: it is what `send_direct` did, and the symptom was a service being dialled +/// once a second for ever with no failures recorded and nothing ever delivered. +/// +/// So the worker rides along, and the caller keeps this alive for as long as it reads. +pub struct HeldResponse { + pub response: hyper::Response, + _worker: Option>, +} + +async fn send( + origin: Origin, + tls: Option>, + mut request: hyper::Request, + config: OutgoingRequestConfig, +) -> std::result::Result { + let OutgoingRequestConfig { + connect_timeout, + first_byte_timeout, + between_bytes_timeout, + .. + } = config; + + // To the origin as admitted, not to whatever the URI's authority spells: the + // two agree by construction, and connecting to the parsed value leaves no + // second interpretation to disagree with. + let host = origin.host.trim_start_matches('[').trim_end_matches(']'); + let tcp = timeout(connect_timeout, TcpStream::connect((host, origin.port))) + .await + .map_err(|_| ErrorCode::ConnectionTimeout)? + .map_err(|e| { + tracing::warn!(%origin, error = %e, "guest egress: could not connect"); + ErrorCode::ConnectionRefused + })?; + + let (mut sender, worker) = if origin.tls { + let config = tls.ok_or(ErrorCode::InternalError(Some( + "no TLS configuration for an https origin".into(), + )))?; + let name = rustls::pki_types::ServerName::try_from(host.to_string()) + .map_err(|_| ErrorCode::HttpRequestUriInvalid)?; + let stream = tokio_rustls::TlsConnector::from(config) + .connect(name, tcp) + .await + .map_err(|e| { + tracing::warn!(%origin, error = %e, "guest egress: TLS failed"); + ErrorCode::TlsProtocolError + })?; + let (sender, conn) = timeout( + connect_timeout, + hyper::client::conn::http1::handshake(TokioIo::new(stream)), + ) + .await + .map_err(|_| ErrorCode::ConnectionTimeout)? + .map_err(hyper_request_error)?; + let worker = wasmtime_wasi::runtime::spawn(async move { + if let Err(e) = conn.await { + tracing::debug!(error = %e, "guest egress connection ended"); + } + }); + (sender, worker) + } else { + let (sender, conn) = timeout( + connect_timeout, + hyper::client::conn::http1::handshake(TokioIo::new(tcp)), + ) + .await + .map_err(|_| ErrorCode::ConnectionTimeout)? + .map_err(hyper_request_error)?; + let worker = wasmtime_wasi::runtime::spawn(async move { + if let Err(e) = conn.await { + tracing::debug!(error = %e, "guest egress connection ended"); + } + }); + (sender, worker) + }; + + // Origin-form on the wire: the scheme and authority are only sent to a proxy. + let path = request + .uri() + .path_and_query() + .map(|p| p.as_str()) + .unwrap_or("/") + .to_string(); + *request.uri_mut() = hyper::Uri::builder() + .path_and_query(path) + .build() + .map_err(|_| ErrorCode::HttpRequestUriInvalid)?; + // HTTP/1.1 requires Host; a guest that did not set one gets the origin's. + if !request.headers().contains_key(hyper::header::HOST) { + let value = if (origin.tls && origin.port == 443) || (!origin.tls && origin.port == 80) { + origin.host.clone() + } else { + format!("{}:{}", origin.host, origin.port) + }; + request.headers_mut().insert( + hyper::header::HOST, + value + .parse() + .map_err(|_| ErrorCode::HttpRequestUriInvalid)?, + ); + } + + let resp = timeout(first_byte_timeout, sender.send_request(request)) + .await + .map_err(|_| ErrorCode::ConnectionReadTimeout)? + .map_err(hyper_request_error)? + .map(|body| body.map_err(hyper_request_error).boxed_unsync()); + + Ok(IncomingResponse { + resp, + worker: Some(worker), + between_bytes_timeout, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use http_body_util::Empty; + use std::time::Duration; + + fn request(uri: &str) -> hyper::Request { + hyper::Request::builder() + .uri(uri) + .body( + Empty::::new() + .map_err(|e| match e {}) + .boxed_unsync(), + ) + .unwrap() + } + + fn config(use_tls: bool) -> OutgoingRequestConfig { + OutgoingRequestConfig { + use_tls, + connect_timeout: Duration::from_secs(2), + first_byte_timeout: Duration::from_secs(2), + between_bytes_timeout: Duration::from_secs(2), + } + } + + #[test] + fn an_origin_is_scheme_host_and_port_and_nothing_else() { + assert_eq!( + Origin::parse("https://ASP.example.com").unwrap(), + Origin { + tls: true, + host: "asp.example.com".into(), + port: 443 + } + ); + assert_eq!( + Origin::parse("http://192.168.127.254:7070/").unwrap(), + Origin { + tls: false, + host: "192.168.127.254".into(), + port: 7070 + } + ); + for bad in [ + "asp.example.com", + "ftp://asp.example.com", + "https://asp.example.com/v1", + "https://asp.example.com?x=1", + "https://user@asp.example.com", + "https://*.example.com", + "https://asp.example.com:0", + "https://asp.example.com:99999", + "https://", + ] { + assert!(Origin::parse(bad).is_err(), "{bad} must be refused"); + } + } + + #[test] + fn only_the_exact_origin_is_admitted() { + let list = + EgressAllowlist::parse(&["https://asp.example.com", "http://127.0.0.1:7070"]).unwrap(); + let admits = |uri: &str, tls: bool| list.admits(&request(uri), &config(tls)).is_some(); + + assert!(admits("https://asp.example.com/v1/info", true)); + assert!(admits("https://asp.example.com:443/v1/info", true)); + assert!(admits("http://127.0.0.1:7070/v1/batch/events", false)); + + assert!(!admits("https://evil.example.com/", true), "another host"); + assert!(!admits("https://sub.asp.example.com/", true), "a subdomain"); + assert!( + !admits("https://asp.example.com:8443/", true), + "another port" + ); + assert!( + !admits("http://asp.example.com/", false), + "plaintext to a TLS origin" + ); + assert!( + !admits("http://127.0.0.1:7071/", false), + "the admin port beside it" + ); + assert!( + !admits("https://asp.example.com/", false), + "the scheme and the guest's TLS flag must agree" + ); + } + + #[test] + fn an_empty_list_admits_nothing() { + let list = EgressAllowlist::parse(&["", " "]).unwrap(); + assert!(list.is_empty()); + assert!(list + .admits(&request("https://example.com/"), &config(true)) + .is_none()); + } + + /// An admitted request really goes out, and the answer really comes back. + #[tokio::test] + async fn an_admitted_request_reaches_the_origin() { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let port = listener.local_addr().unwrap().port(); + let server = tokio::spawn(async move { + let (mut socket, _) = listener.accept().await.unwrap(); + let mut buf = vec![0u8; 4096]; + let n = socket.read(&mut buf).await.unwrap(); + let head = String::from_utf8_lossy(&buf[..n]).to_string(); + socket + .write_all(b"HTTP/1.1 200 OK\r\ncontent-length: 5\r\n\r\nhello") + .await + .unwrap(); + head + }); + + let list = EgressAllowlist::parse(&[format!("http://127.0.0.1:{port}")]).unwrap(); + let req = request(&format!("http://127.0.0.1:{port}/v1/info?x=1")); + let origin = list.admits(&req, &config(false)).expect("admitted"); + let response = send(origin, None, req, config(false)).await.expect("sent"); + assert_eq!(response.resp.status(), 200); + let body = response + .resp + .into_body() + .collect() + .await + .unwrap() + .to_bytes(); + assert_eq!(&body[..], b"hello"); + + let head = server.await.unwrap(); + assert!( + head.starts_with("GET /v1/info?x=1 HTTP/1.1"), + "origin-form: {head}" + ); + assert!( + head.to_ascii_lowercase() + .contains(&format!("host: 127.0.0.1:{port}")), + "{head}" + ); + } +} diff --git a/runtime/src/serve/http.rs b/runtime/src/serve/http.rs new file mode 100644 index 0000000..21606a2 --- /dev/null +++ b/runtime/src/serve/http.rs @@ -0,0 +1,1881 @@ +//! Dispatching an incoming HTTP request into a `wasi:http/proxy` guest. +//! +//! **One instance per request. No two requests ever share one.** +//! +//! That is the invariant this module exists to hold, and it is where the +//! boundary between two clients actually lives. A `Store` is what separates two +//! wasm instances; give two clients one instance and the separation becomes +//! guest code instead, so a single bug that confuses two clients stops being a +//! leak of one and becomes a total compromise. A trap has nothing to poison and +//! leaked resource handles die with the store, both for the same reason. +//! +//! What a guest keeps therefore has to reach the filesystem, because there is +//! nothing else: no memory outlives the call that created it. + +use std::net::SocketAddr; +use std::sync::Arc; +use std::time::Duration; + +use anyhow::{Context, Result}; +use tokio::net::TcpListener; +use wasmtime::component::{Component, InstancePre}; +use wasmtime::{Engine, Store, UpdateDeadline}; +use wasmtime_wasi_http::io::TokioIo; +use wasmtime_wasi_http::p2::bindings::http::types::Scheme; +use wasmtime_wasi_http::p2::bindings::ProxyPre; +use wasmtime_wasi_http::p2::body::HyperOutgoingBody; +use wasmtime_wasi_http::p2::WasiHttpView; + +use crate::auth::{AuthEndpoints, Gate}; +use crate::linker::build_linker; +use crate::run::GuestEnvironment; +use crate::serve::acme::CertificateSlot; +use crate::serve::attest::{nonce_from_headers, ResponseAttestor}; +use crate::serve::client::apply_tenant; +use crate::serve::pool::{LiveTenant, PoolLimits, TenantPool}; +use crate::serve::progress::{Counting, StreamProgress}; +use crate::state::State; +use crate::tenant::tenant_root_by_id; +use nitro_nsm::Nsm; + +/// How the guest is served. +#[derive(Debug)] +pub struct ServeConfig { + /// Opt-in standing authorization for tenant-bound durable tasks. + pub background_tasks: Option, + /// Opt-in push notifications. The registry and the forwarder are opened + /// here rather than by the caller, because both need the mounted + /// filesystem and the trusted clock, and this is where those exist. + pub notify: Option, + /// Address to accept on. + pub addr: SocketAddr, + /// Where TLS connections take their serving identity. `None` serves + /// plaintext. + /// + /// A renewal is a `set()` on it. Connections already established keep the + /// identity they loaded, which is the property per-response attestation + /// depends on and the only way to exercise it without driving a real ACME + /// order. + pub certificate: Option, + /// A running ACME client, instead of a fixed certificate. + pub acme: Option, + /// NSM to sign the attestation document on each `/auth/*` response. + /// + /// Only useful alongside `tls`: the document binds the serving + /// certificate, and without one there is nothing to bind. Supplying it + /// without TLS is refused rather than silently serving documents that + /// promise a binding they do not have. + pub attestation: Option>, + /// How long a guest may take to produce a response head. + /// + /// A guest that neither returns nor sets a response otherwise hangs that + /// tenant forever — and only that tenant, since nothing else queues behind + /// it. See [`EPOCH_TICK`]. + pub request_timeout: Duration, + /// How long an interaction may run once it has answered. + /// + /// Deliberately separate from `request_timeout`: that one bounds getting a + /// head out, and does not apply afterwards, so without this a stream had no + /// ceiling at all and held its tenant's slot indefinitely. + pub max_interaction: Duration, + /// Give each authenticated client their own filesystem. `None` keeps the + /// original model: one filesystem, a fresh instance per request. + pub tenancy: Option>, + /// `/auth/*` and the gate every other request must pass. + /// + /// `None` serves the guest to anyone who can open a connection. That is a + /// development and QEMU arrangement, and [`serve_component`] says so + /// loudly at startup rather than leaving it to be noticed. + pub authentication: Option<(Arc, Arc)>, + /// What guests may reach over `wasi:http`. [`EgressPolicy::Denied`] unless + /// the deployment names origins. + pub egress: crate::serve::EgressPolicy, +} + +impl Default for ServeConfig { + /// One request at a time. + /// + /// The store layer is safe to share — `Fs` keeps its handles under a + /// `parking_lot::Mutex` and its transaction state under a `tokio::Mutex`, + /// so nothing corrupts. The *guest* is the problem. `docs/COMPATIBILITY.md` + /// records that SQLite on WASI has to run `locking_mode=EXCLUSIVE`, because + /// WASI has no `fcntl` and therefore no file locking; two guest instances + /// holding the same database would each believe they had it alone. + /// + /// A guest that keeps no cross-request state in the filesystem can raise + /// this safely. One that opens a database cannot, and would fail in a way + /// that looks like corruption rather than contention — so the default is + /// the one that cannot surprise anybody. + fn default() -> Self { + ServeConfig { + background_tasks: None, + notify: None, + addr: ([0, 0, 0, 0], 8080).into(), + certificate: None, + acme: None, + attestation: None, + request_timeout: DEFAULT_REQUEST_TIMEOUT, + max_interaction: DEFAULT_MAX_INTERACTION, + // Off. Turning it on changes what two clients may do at the same + // time, which a guest may have been relying on — so it is asked + // for, never inherited. + tenancy: None, + authentication: None, + egress: crate::serve::EgressPolicy::Denied, + } + } +} + +/// How often the epoch advances, and therefore the resolution of the deadline +/// a spinning guest is trapped by. +/// +/// Coarse on purpose: the ticker wakes on every interval for the life of the +/// process, and the deadline only needs to be accurate to a fraction of the +/// request timeout. +const EPOCH_TICK: Duration = Duration::from_millis(250); + +/// Consecutive silent epoch expiries before a call is judged a runaway. +/// +/// This — not `request_timeout` — is what bounds how long a guest can burn a +/// worker after its budget is spent, and it is deliberately independent of the +/// budget: a deployment that allows generous head timeouts should not thereby +/// allow generous spinning. Eight ticks is two seconds, long enough that a +/// loaded machine delivering frames late is never mistaken for one delivering +/// nothing, short enough that a genuine runaway is gone before it matters. +const SILENT_TICKS_BEFORE_TRAP: u32 = 8; + +/// Default ceiling on producing a response head. +const DEFAULT_REQUEST_TIMEOUT: Duration = Duration::from_secs(30); + +/// How long an interaction may run once it has produced a head. +/// +/// Minutes rather than seconds, because a signing session legitimately takes +/// them and a stream doing its job must not be cut off for being slow. What +/// this bounds is the other case: a peer that opened an interaction and then +/// went quiet still holds its tenant's only slot, and without a ceiling it +/// holds it until the process ends. +const DEFAULT_MAX_INTERACTION: Duration = Duration::from_secs(300); + +/// A guest instance and the store it is bound to. +/// +/// The two travel together because `instantiate_async` binds a `Proxy` to one +/// store: separating them would produce a handle that looks usable and +/// resolves against the wrong memory. +pub struct GuestInstance { + store: Store, + proxy: wasmtime_wasi_http::p2::bindings::Proxy, + /// Shared with this store's epoch callback, which was installed once and + /// outlives every request the instance serves. Reset per call rather than + /// replaced, so the callback never holds a stale one. + progress: Arc, +} + +/// A compiled guest plus the environment its instances are built from. +/// +/// Compilation and `instantiate_pre` happen once, at startup: per-request +/// instantiation is then cheap, and — more usefully inside an enclave — a +/// component that will not compile fails at boot rather than on the first +/// request to arrive. +pub struct ServeHandle { + pre: ProxyPre, + background_pre: InstancePre, + tasks: Option>, + /// Connections this runtime holds for guests — see [`crate::stream`]. + streams: Option>, + notify: Option>, + egress: crate::serve::EgressPolicy, + guest: Arc, + /// One request at a time for callers with no resolved tenant. + /// + /// There is no identity to separate them by, so they are treated as one: + /// anonymous requests serialise against each other and against nothing + /// else. Unbounded would let anyone with a socket multiply wasm linear + /// memories, which is the same exhaustion a per-tenant lock prevents for + /// everybody who *is* identified. + anonymous: Arc>, + /// Ceiling on producing a response head. See [`ServeHandle::watchdog`]. + timeout: Duration, + /// How long a call may run after its head is out. See `supervise`. + max_interaction: Duration, + /// Per-client filesystems, when the deployment asked for them. + /// + /// `None` is the original model and stays the default: one filesystem, a + /// fresh instance per request, every request serialised against every + /// other. Turning this on changes what two clients can do at the same time, + /// which is a property a guest may have been relying on — so it is opted + /// into, and PCR0 records the choice. + tenancy: Option>, +} + +/// Everything needed to give a client their own corner of the filesystem. +/// +/// No backends and no mounting: there is one filesystem, and a client costs a +/// directory in it. What is per-client is the *view* — which directory the +/// guest calls `/` — plus a warm instance and a lock. +pub struct Tenancy { + pool: TenantPool, +} + +impl std::fmt::Debug for Tenancy { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("Tenancy") + .field("tenants", &self.pool.len()) + .finish_non_exhaustive() + } +} + +impl Tenancy { + pub fn new(limits: PoolLimits) -> Self { + Tenancy { + pool: TenantPool::new(limits), + } + } + + pub fn pool(&self) -> &TenantPool { + &self.pool + } +} + +impl ServeHandle { + /// An `Engine` configured for the watchdog, plus the ticker that drives it. + /// + /// Epoch interruption is what makes a spinning guest interruptible at all: + /// without it, wasm that never yields cannot be stopped, and + /// `tokio::task::abort` has no await point to act on. A dedicated thread + /// advances the epoch independently of Tokio workers and exits when the + /// last engine handle is dropped. + pub fn engine_with_watchdog() -> Result { + let mut config = wasmtime::Config::new(); + config.epoch_interruption(true); + let engine = Engine::new(&config)?; + + let ticker = engine.weak(); + // The ticker must run even when guest CPU loops occupy every Tokio + // worker. An async ticker on that same executor cannot guarantee it. + std::thread::Builder::new() + .name("guest-epoch".into()) + .spawn(move || { + loop { + std::thread::sleep(EPOCH_TICK); + // A weak handle, so this thread cannot keep a dropped engine + // alive — it simply stops when the engine goes. + match ticker.upgrade() { + Some(engine) => engine.increment_epoch(), + None => break, + } + } + })?; + Ok(engine) + } + + pub fn new(engine: &Engine, component_bytes: &[u8], guest: GuestEnvironment) -> Result { + let component = Component::new(engine, component_bytes) + .map_err(|e| anyhow::anyhow!(e.to_string())) + .context("compiling guest component")?; + + let mut linker = build_linker(engine).map_err(|e| anyhow::anyhow!(e.to_string()))?; + wasmtime_wasi_http::p2::add_only_http_to_linker_async(&mut linker) + .map_err(|e| anyhow::anyhow!(e.to_string())) + .context("adding wasi:http to the linker")?; + + let instance_pre: InstancePre = linker + .instantiate_pre(&component) + .map_err(|e| anyhow::anyhow!(e.to_string())) + .context("pre-instantiating guest component")?; + let pre = ProxyPre::new(instance_pre.clone()) + .map_err(|e| anyhow::anyhow!(e.to_string())) + .context("guest does not export wasi:http/incoming-handler")?; + + Ok(ServeHandle { + background_pre: instance_pre, + tasks: None, + streams: None, + notify: None, + egress: crate::serve::EgressPolicy::Denied, + pre, + guest: Arc::new(guest), + anonymous: Arc::new(tokio::sync::Mutex::new(())), + timeout: DEFAULT_REQUEST_TIMEOUT, + max_interaction: DEFAULT_MAX_INTERACTION, + tenancy: None, + }) + } + + /// The environment instances are built from — the runtime's own + /// filesystem, the clock, and the entropy source. + pub fn environment(&self) -> &Arc { + &self.guest + } + + /// Give each authenticated client their own filesystem and warm instance. + pub fn with_tenancy(mut self, tenancy: Arc) -> Self { + self.tenancy = Some(tenancy); + self + } + + pub fn with_tasks(mut self, queue: Arc) -> Result { + anyhow::ensure!( + self.tenancy.is_some(), + "background tasks require tenant isolation" + ); + self.tasks = Some(queue); + Ok(self) + } + + /// Let the runtime hold outbound connections on a guest's behalf. + /// + /// Tenant isolation is required for the same reason tasks require it: a + /// connection belongs to a tenant, and the invocations its messages cause + /// run in that tenant's filesystem. Without tenancy there is nobody to + /// attribute either to. + pub fn with_streams(mut self, registry: Arc) -> Result { + anyhow::ensure!( + self.tenancy.is_some(), + "held connections require tenant isolation" + ); + self.streams = Some(registry); + Ok(self) + } + + /// Let the guest wake its tenant's devices. + /// + /// Tenant isolation is required for the same reason tasks require it: a + /// device belongs to a tenant, and without tenancy there is no tenant to + /// bind an enrolment to. + pub fn with_notify(mut self, notifier: Arc) -> Result { + anyhow::ensure!( + self.tenancy.is_some(), + "notifications require tenant isolation" + ); + self.notify = Some(notifier); + Ok(self) + } + + /// Let guests reach the origins [`crate::serve::EgressPolicy`] names — for + /// requests and background tasks alike, since renewing something on a + /// schedule is exactly the work that happens with nobody connected. + pub fn with_egress(mut self, egress: crate::serve::EgressPolicy) -> Self { + self.egress = egress; + self + } + + /// An internal component export, never an HTTP route or a fabricated token. + /// One message from a held connection, as one guest invocation. + /// + /// The same shape as [`Self::run_background`] and for the same reasons: a + /// fresh instance, the tenant's lock, a wall deadline. What differs is only + /// what woke it — a counterparty rather than a clock — and that it may reply. + /// + /// `interactive` is FALSE. A message arriving over a connection is not its + /// owner asking for something, so it cannot open connections or schedule + /// work, exactly as a background run cannot. + pub(crate) async fn run_message( + &self, + streams: &Arc, + tenant: [u8; 16], + id: &str, + message_id: &str, + payload: Vec, + ) -> Result> { + let tenancy = self + .tenancy + .as_ref() + .context("held connections require tenants")?; + let checkout = tenancy.pool.checkout(&tenant); + let mut guard = checkout.slot().tenant().clone().lock_owned().await; + if let Some(t) = guard.as_mut() { + t.instance = None; + } + let work = async { + let scope = crate::tenant::existing_tenant_root(self.guest.fs(), tenant).await?; + let state = self.guest.new_state_scoped(scope)?; + let mut store = Store::new(self.pre.engine(), state); + store.set_epoch_deadline(1); + store.epoch_deadline_async_yield_and_update(1); + let instance = self.background_pre.instantiate_async(&mut store).await?; + store.data_mut().streams = Some(crate::stream::StreamContext { + registry: streams.clone(), + tenant, + interactive: false, + }); + store.data_mut().set_egress(self.egress.clone()); + store.data_mut().notify = + self.notify + .as_ref() + .map(|notifier| crate::notify::NotifyContext { + notifier: notifier.clone(), + tenant, + interactive: false, + }); + let run = instance + .get_typed_func::<(String, String, Vec), (std::result::Result, String>,)>( + &mut store, + "on-message", + ) + .map_err(|e| anyhow::anyhow!(e.to_string())) + .context("held connections require the on-message export")?; + let (result,) = run + .call_async( + &mut store, + (id.to_string(), message_id.to_string(), payload), + ) + .await?; + let bytes = result.map_err(anyhow::Error::msg)?; + anyhow::ensure!( + bytes.len() <= crate::stream::MAX_MESSAGE, + "reply exceeds the message ceiling" + ); + Ok(bytes) + }; + // The same wall clock that bounds an interaction. A guest parked in a + // host call executes no wasm, so the epoch alone would never reach it. + tokio::time::timeout(self.timeout, work) + .await + .context("the guest took too long to answer a message")? + } + + /// A fresh instance shares the tenant lock; drop its warm HTTP instance so + /// no cached database handles survive a background mutation. + pub(crate) async fn run_background( + &self, + queue: &Arc, + task: &crate::tasks::Task, + ) -> Result>> { + let tenancy = self + .tenancy + .as_ref() + .context("background tasks require tenants")?; + let checkout = tenancy.pool.checkout(&task.tenant); + let Ok(mut guard) = checkout.slot().tenant().clone().try_lock_owned() else { + return Ok(None); + }; + if !queue.begin(task).await? { + return Ok(Some(Vec::new())); + } + if let Some(tenant) = guard.as_mut() { + tenant.instance = None; + } + let work = async { + let scope = crate::tenant::existing_tenant_root(self.guest.fs(), task.tenant).await?; + let state = self.guest.new_state_scoped(scope)?; + let mut store = Store::new(self.pre.engine(), state); + // Yield every tick, even in CPU-only loops, so the wall deadline + // can cancel both guest code and async host calls. + store.set_epoch_deadline(1); + store.epoch_deadline_async_yield_and_update(1); + let instance = self.background_pre.instantiate_async(&mut store).await?; + store.data_mut().tasks = Some(crate::tasks::TaskContext { + queue: queue.clone(), + tenant: task.tenant, + interactive: false, + }); + store.data_mut().set_egress(self.egress.clone()); + store.data_mut().notify = + self.notify + .as_ref() + .map(|notifier| crate::notify::NotifyContext { + notifier: notifier.clone(), + tenant: task.tenant, + interactive: false, + }); + let run = instance + .get_typed_func::<(String, Vec), (std::result::Result, String>,)>( + &mut store, "run-task", + )?; + let (result,) = run + .call_async(&mut store, (task.run_id(), task.payload.clone())) + .await?; + let bytes = result.map_err(anyhow::Error::msg)?; + anyhow::ensure!(bytes.len() <= 64 * 1024, "task result exceeds 64 KiB"); + Ok(Some(bytes)) + }; + tokio::time::timeout(queue.limits.timeout, work) + .await + .context("background task deadline exceeded")? + } + + /// Override the response-head deadline. + pub fn with_timeout(mut self, timeout: Duration) -> Self { + self.timeout = timeout; + self + } + + /// Override how long an interaction may run once it has answered. + /// + /// Distinct from [`ServeHandle::with_timeout`], and the distinction is the + /// point: one bounds how long a guest may take to *start* answering, the + /// other how long it may go on. A stream that is healthy for minutes is + /// normal; a tenant's slot held for hours is not. + pub fn with_max_interaction(mut self, max_interaction: Duration) -> Self { + self.max_interaction = max_interaction; + self + } + + /// Ticks of [`EPOCH_TICK`] a guest gets before the epoch traps it. + /// + /// Deliberately **shorter** than the timeout, and that ordering is the + /// whole mechanism. The epoch is the only thing that can stop wasm which + /// never yields; `tokio::task::abort` needs an await point and a spinning + /// guest never reaches one. Give the deadline slack *past* the timeout — + /// as a first attempt here did, to get a tidier error message — and the + /// abort fires first against a guest nothing can interrupt, which leaves + /// the task spinning for the life of the process and hangs runtime + /// shutdown. The nicer message cost the entire guarantee. + /// + /// So the epoch fires first and the guest traps with a real reason. The + /// timeout stays as the backstop for what the epoch cannot see: a guest + /// parked in a *host* call, where there is no wasm executing to interrupt + /// but there is an await point for `abort` to land on. + fn watchdog(&self) -> u64 { + let ticks = self.timeout.as_millis() / EPOCH_TICK.as_millis(); + (ticks.saturating_sub(2)).max(1) as u64 + } + + /// Run one request, in an instance built for it and discarded after it. + /// + /// `scheme` is what the *client* used, which is not what this function was + /// reached over: with TLS terminated upstream of here the connection is + /// plaintext, but the guest must be told `https` or it will build wrong + /// absolute URLs and set wrong cookie flags. + /// + /// Generic over the request body rather than taking + /// `hyper::body::Incoming`, which cannot be constructed outside a real + /// connection — the whole dispatch path would be untestable without a + /// listening socket. + /// + /// `tenant` is required rather than defaulted, so every caller has to + /// answer the question. It is **not** a way around the gate: with one + /// configured, [`serve_connection`] verifies an assertion before it gets + /// here and there is no path that reaches a guest without one. + pub async fn handle( + &self, + scheme: Scheme, + mut req: hyper::Request, + tenant: Option<&[u8; 16]>, + ) -> Result> + where + B: hyper::body::Body + Send + Unpin + 'static, + B::Error: Into, + { + // Before anything else touches the request, and before it is handed to + // `new_incoming_request`, which copies the header map into the guest + // verbatim. This is the only ingress, so this is the only place the + // header can be made trustworthy. + apply_tenant(tenant, &mut req); + + // Concurrency is per tenant, and the lock that enforces it is taken + // inside each arm — the per-tenant mutex for an identified caller, one + // shared lock for the rest. There is no global ceiling: two tenants + // have nothing to queue on, which is the whole point of giving them + // separate directories and separate instances. + // + // Either lock is held until the guest task finishes, body included, so + // a tenant's second request waits for its first to be genuinely done + // rather than merely to have produced headers. + match (&self.tenancy, tenant) { + (Some(tenancy), Some(tenant)) => { + self.in_tenant(tenancy.clone(), *tenant, scheme, req).await + } + // No tenant resolved. Either the deployment configured no gate, or + // the caller reached a path that does not identify itself — and + // with no identity there is nothing to separate them by, so they + // share one lock and one filesystem: the runtime's own. + _ => self.in_fresh_instance(scheme, req).await, + } + } + + /// One request, in this client's own filesystem and warm instance. + /// + /// The lock is held for the whole call — including the body, because + /// `call_handle` does not return until the guest has finished writing it — + /// and it is what serialises this client against **itself and nothing + /// else**. Two clients holding two different locks over two different + /// filesystems is the entire point: their commits do not queue, because + /// there is nothing for them to queue on. + async fn in_tenant( + &self, + tenancy: Arc, + tenant_id: [u8; 16], + scheme: Scheme, + req: hyper::Request, + ) -> Result> + where + B: hyper::body::Body + Send + Unpin + 'static, + B::Error: Into, + { + // Checked out before the lock is taken, so the eviction sweep can see + // this tenant is busy and leave it alone. The guard releases that mark + // however the request ends. + let checkout = tenancy.pool.checkout(&tenant_id); + // Owned, so it can move into the task below and outlive this function. + let mut guard = checkout.slot().tenant().clone().lock_owned().await; + + if guard.is_none() { + // First use of this slot, under the lock — so two simultaneous + // first requests from one client resolve to one directory, the + // second waiting here rather than racing the first. + let opened = tenant_root_by_id(self.guest.fs(), tenant_id).await?; + tracing::info!( + tenant = %hex::encode(opened.tenant_id), + arrival = ?opened.arrival, + "tenant directory ready" + ); + *guard = Some(LiveTenant { + scope: opened.scope, + tenant_id: opened.tenant_id, + instance: None, + requests: 0, + }); + } + + let tenant = guard.as_mut().expect("just built"); + // Rebuilt when absent, and when this one has served long enough: wasm + // linear memory never shrinks, so an instance that lived forever would + // only grow. Cheap — ~24 µs, and no I/O, because there is no + // filesystem to bring up with it. + let stale = tenant.requests >= tenancy.pool.limits().max_requests_per_instance; + if tenant.instance.is_none() || stale { + tenant.instance = Some(self.instantiate(tenant.scope.clone()).await?); + tenant.requests = 0; + } + tenant.requests += 1; + + let instance = tenant.instance.as_mut().expect("just built"); + instance.store.data_mut().streams = + self.streams + .as_ref() + .map(|registry| crate::stream::StreamContext { + registry: registry.clone(), + tenant: tenant_id, + // Its owner is here, so this invocation may open and close + // connections — unlike one caused by a message arriving on + // one, which may only reply. + interactive: true, + }); + instance.store.data_mut().tasks = + self.tasks.as_ref().map(|queue| crate::tasks::TaskContext { + queue: queue.clone(), + tenant: tenant_id, + interactive: true, + }); + instance.store.data_mut().set_egress(self.egress.clone()); + instance.store.data_mut().notify = + self.notify + .as_ref() + .map(|notifier| crate::notify::NotifyContext { + notifier: notifier.clone(), + tenant: tenant_id, + interactive: true, + }); + // Reset every request: the epoch deadline is absolute, so a reused + // store would otherwise inherit whatever the last request left. The + // same is true of the progress this call will be judged on. + instance.store.set_epoch_deadline(self.watchdog()); + instance.progress.reset(); + let progress = instance.progress.clone(); + let (sender, receiver) = tokio::sync::oneshot::channel(); + // Wrapped before it becomes a guest resource, which is the last point + // the runtime holds it. + let req = req.map(|body| Counting::new(body, progress.clone())); + let req = instance + .store + .data_mut() + .http() + .new_incoming_request(scheme, req)?; + let out = instance + .store + .data_mut() + .http() + .new_response_outparam(sender)?; + + let task = tokio::task::spawn(async move { + // Moved in so the lock outlives the response head. A pooled store + // must not be handed to this tenant's next request until the call + // has finished writing its body — which is also what makes "one + // active request per tenant" true rather than "one set of headers + // at a time". Released when this future completes, fails, or is + // aborted. + let mut guard = guard; + let _checkout = checkout; + let tenant = guard.as_mut().expect("held across the call"); + // Taken out of the slot for the duration, and put back only on a + // clean finish. Held *by reference* instead — as this once did — + // and an aborted call leaves the instance where it sits: `abort` + // drops this future at the await below, so nothing after it runs, + // and the next request for this tenant re-enters a store whose + // `call_handle` was cancelled part-way through. Ownership is what + // makes "an interrupted instance is never reused" true for the + // abort path and not only for the trap path, because a dropped + // future drops what it owns. + let mut instance = tenant.instance.take().expect("built before the call"); + let result = instance + .proxy + .wasi_http_incoming_handler() + .call_handle(&mut instance.store, req, out) + .await; + + // One client's instance, and nothing else. The filesystem is + // shared and untouched; the next caller for this client gets a + // fresh instance over the same directory. + if result.is_err() { + tracing::warn!("guest trapped; this client's instance will be rebuilt"); + } else if !instance.store.data().resources_settled() { + tracing::debug!("guest left resources behind; rebuilding its instance"); + } else { + tenant.instance = Some(instance); + } + result + }); + + await_head(self.timeout, self.max_interaction, task, receiver, progress).await + } + + /// Build an instance whose guest sees `scope` as `/`. + /// + /// The one place a store and an instance are made, so the per-request path + /// and the per-tenant path cannot drift in what a guest is handed. + async fn instantiate(&self, scope: Arc) -> Result { + let mut store = Store::new(self.pre.engine(), self.guest.new_state_scoped(scope)?); + let progress = Arc::new(StreamProgress::new()); + store.set_epoch_deadline(self.watchdog()); + self.arm_watchdog(&mut store, progress.clone()); + let proxy = self + .pre + .instantiate_async(&mut store) + .await + .map_err(|e| anyhow::anyhow!(e.to_string())) + .context("instantiating the guest")?; + Ok(GuestInstance { + store, + proxy, + progress, + }) + } + + /// Let a call that is moving bytes outlive the budget; keep one that is not + /// on exactly the schedule it had before. + /// + /// The deadline set beside this is what a request/response call gets, and + /// for anything that finishes inside the timeout nothing here ever runs. + /// When it does expire, the question stops being "how long has this taken" + /// and becomes "is this still a conversation" — because a signing session + /// legitimately open for minutes and a guest spinning in a loop are + /// indistinguishable by elapsed time and obvious by traffic. + /// + /// `Yield` rather than `Continue`: extending alone leaves a guest that + /// makes progress *and* spins between frames with no await point for + /// `await_head`'s abort to land on. Yielding creates one every tick, which + /// costs a reschedule per [`EPOCH_TICK`] and keeps the abort path real. + /// + /// Note what this deliberately cannot see: a guest parked in a *host* call + /// executes no wasm, so no epoch check is reached and nothing here fires. + /// That gap is the idle supervisor's, not this one's. + fn arm_watchdog(&self, store: &mut Store, progress: Arc) { + let silent = EPOCH_TICK * SILENT_TICKS_BEFORE_TRAP; + store.epoch_deadline_callback(move |_| { + // A single tick at a time once the budget is spent, so the + // question gets asked again promptly rather than handing out + // another full budget on one frame of evidence. + if progress.still_working(SILENT_TICKS_BEFORE_TRAP) { + Ok(UpdateDeadline::Yield(1)) + } else { + Err(wasmtime::Error::msg(format!( + "the guest ran {silent:?} past its budget without moving a byte \ + in either direction" + ))) + } + }); + } + + /// The only dispatch path. A fresh store, a fresh instance, both dropped + /// when the request ends — so nothing survives but what was committed. + async fn in_fresh_instance( + &self, + scheme: Scheme, + req: hyper::Request, + ) -> Result> + where + B: hyper::body::Body + Send + Unpin + 'static, + B::Error: Into, + { + // Anonymous callers are one identity, so they queue behind each other. + // Taken before the store is built: an instance is a wasm linear memory, + // and building one per waiting request is the exhaustion this prevents. + let anonymous = self.anonymous.clone().lock_owned().await; + + let mut store = Store::new(self.pre.engine(), self.guest.new_state()?); + let progress = Arc::new(StreamProgress::new()); + store.set_epoch_deadline(self.watchdog()); + self.arm_watchdog(&mut store, progress.clone()); + let (sender, receiver) = tokio::sync::oneshot::channel(); + let req = req.map(|body| Counting::new(body, progress.clone())); + let req = store.data_mut().http().new_incoming_request(scheme, req)?; + let out = store.data_mut().http().new_response_outparam(sender)?; + let pre = self.pre.clone(); + + // The guest runs in its own task so it can keep streaming a body after + // the status line and headers have gone out. The lock goes with it, so + // it is held for the whole call rather than released at the head. + let task = tokio::task::spawn(async move { + let _anonymous = anonymous; + let proxy = pre.instantiate_async(&mut store).await?; + proxy + .wasi_http_incoming_handler() + .call_handle(store, req, out) + .await + }); + + await_head(self.timeout, self.max_interaction, task, receiver, progress).await + } + + /// Prove the guest instantiates, before a single request depends on it. + /// + /// Builds one instance and drops it. Narrower than it looks, and worth + /// saying so: `Component::new` and `instantiate_pre` already fail at + /// startup for a guest that will not compile or whose imports do not + /// resolve. What this adds is the *initialiser* — a guest that traps or + /// hangs in `start` would otherwise answer 500 to every request while the + /// enclave looked perfectly healthy from the outside. + /// + /// The cost is running that initialiser once more than strictly necessary, + /// which is nothing new: [`ServeHandle::in_fresh_instance`] runs it on + /// every request already. + pub async fn verify_instantiates(&self) -> Result<()> { + let mut store = Store::new(self.pre.engine(), self.guest.new_state()?); + // A deadline of its own, or a guest that hangs in `start` would hang + // the boot rather than failing it. + store.set_epoch_deadline(self.watchdog()); + self.pre + .instantiate_async(&mut store) + .await + .map_err(|e| anyhow::anyhow!(e.to_string())) + .context("instantiating the guest")?; + if self.tasks.is_some() { + let instance = self.background_pre.instantiate_async(&mut store).await?; + instance + .get_typed_func::<(String, Vec), (std::result::Result, String>,)>( + &mut store, "run-task", + ) + .map_err(|e| anyhow::anyhow!(e.to_string())) + .context("background tasks require the run-task export")?; + } + Ok(()) + // The store and the instance drop here. Nothing is kept: an instance + // that outlived this call would be an instance two requests could + // share, which is the whole thing this design refuses. + } +} + +/// Wait for the response head, and leave the guest task to finish the body. +/// +/// Returns as soon as the guest has set a response. Deliberately: the body may +/// still be streaming out of the task, and hyper cannot read it until this +/// returns — waiting for the task here would deadlock body delivery. +/// +/// **The permit is not here.** It lives in the spawned task, so a slot is freed +/// when the guest is genuinely finished — or when its task fails or is aborted +/// — rather than when its headers happened to appear. A free function rather +/// than a method because that lifetime is worth testing on its own, without an +/// engine and a compiled component to build a `ServeHandle` around. +/// Bound how long an interaction may run, once its head is out. +/// +/// Until this existed nothing watched the call after `await_head` returned: the +/// `JoinHandle` was dropped, which detaches, and the only remaining limits were +/// the epoch — which a guest parked in a host call never reaches, because it +/// executes no wasm — and `wasi:http`'s hardcoded ten-minute between-bytes +/// ceiling. A client that opened a stream and then said nothing held its +/// tenant's single slot for as long as it liked. +/// +/// A wall clock is the right instrument here precisely where the epoch is the +/// wrong one. A guest blocked in a host call *is* at an await point inside +/// `call_handle`, so `abort` reaches it — and dropping the task drops the +/// tenant's guard, the pool checkout, and the instance, none of which it can +/// hand back. +/// +/// Detached deliberately: the caller is returning a response body that hyper is +/// about to stream, so it cannot hold this. The handle races the deadline and +/// goes away when either finishes. +fn supervise(max_interaction: Duration, task: tokio::task::JoinHandle>) { + tokio::task::spawn(async move { + let mut task = task; + if tokio::time::timeout(max_interaction, &mut task) + .await + .is_err() + { + tracing::info!( + limit = ?max_interaction, + "an interaction reached its deadline; abandoning it and freeing its tenant" + ); + task.abort(); + } + }); +} + +async fn await_head( + timeout: Duration, + max_interaction: Duration, + task: tokio::task::JoinHandle>, + receiver: tokio::sync::oneshot::Receiver< + std::result::Result< + hyper::Response, + wasmtime_wasi_http::p2::bindings::http::types::ErrorCode, + >, + >, + progress: Arc, +) -> Result> { + let waited = match tokio::time::timeout(timeout, receiver).await { + Ok(waited) => waited, + Err(_) => { + // Nothing arrived in time. The epoch should already have + // trapped a spinning guest — see `watchdog` — so reaching here + // means the guest is parked somewhere the epoch cannot see, + // which is exactly where `abort` does work. + // Aborting drops the task's locals, and the permit is one of + // them — so an abandoned guest frees its slot rather than + // holding it for the life of the process. + task.abort(); + anyhow::bail!("guest did not produce a response within {timeout:?}; abandoned"); + } + }; + + match waited { + // Wrapped on the way out, which is the first point the runtime holds + // it again: from here the count follows what hyper actually drains, + // so a guest writing into a buffer nobody reads earns no extension. + Ok(Ok(resp)) => { + // From here `await_head` is out of the picture, so the watchdog may + // stop racing it and start judging silence on its merits. + progress.head_sent(); + supervise(max_interaction, task); + Ok(resp.map(|body| { + use http_body_util::BodyExt; + Counting::new(body, progress).boxed_unsync() + })) + } + Ok(Err(e)) => Err(e.into()), + // The sender dropped with the `Store`, so the guest returned or + // trapped without setting a response. Whatever the task says is + // the real error; "never set a response" alone would send someone + // looking in the wrong place. + Err(_) => { + match task.await { + Ok(Ok(())) => { + anyhow::bail!("guest returned without calling response-outparam::set") + } + Ok(Err(e)) => Err(anyhow::anyhow!(e.to_string()) + .context("guest failed before setting a response")), + Err(e) => Err(anyhow::Error::from(e).context("guest task panicked")), + } + } + } +} + +/// Refuse a resumed session, and say so. +/// +/// Serving configs disable resumption (see `serve::tls`), so this should be +/// unreachable — it is here because the alternative to being unreachable is +/// attesting a certificate the connection was never authenticated under, and +/// that failure would be silent. A loud refusal is the cheaper mistake. +fn resumed(stream: &tokio_rustls::server::TlsStream, peer: SocketAddr) -> bool { + let (_, connection) = stream.get_ref(); + if connection.handshake_kind() == Some(rustls::HandshakeKind::Resumed) { + tracing::error!( + %peer, + "refusing a resumed TLS session: the certificate it was authenticated \ + under cannot be established, so nothing about it can be attested" + ); + return true; + } + false +} + +/// A fixed body, in the shape the guest dispatch wants. +fn full(bytes: bytes::Bytes) -> HyperOutgoingBody { + use http_body_util::BodyExt; + http_body_util::Full::new(bytes) + .map_err(|e: std::convert::Infallible| match e {}) + .boxed_unsync() +} + +/// A refusal, in one sentence and no detail. +fn refused(status: hyper::StatusCode, detail: &str) -> hyper::Response { + hyper::Response::builder() + .status(status) + .header("content-type", "application/json") + .body(full(bytes::Bytes::from( + serde_json::json!({ "error": detail }).to_string(), + ))) + .expect("response is well formed") +} + +/// Serve one connection, whatever it is wrapped in. +/// +/// Generic over the stream so the three arms above — fixed TLS, ACME TLS and +/// plaintext — share one body instead of three copies of it. All three satisfy +/// the bound, so this monomorphises and costs nothing at runtime. +/// +/// The service closure is built here rather than by the caller because it must +/// capture `client`, and `client` does not exist until the handshake has +/// completed. +/// Everything a connection needs that is the same for every connection. +/// +/// Grouped because the per-connection values — the certificate this handshake +/// presented — are the interesting ones, and a signature where they are lost +/// among six process-wide `Arc`s hides that. +#[derive(Clone)] +pub struct Routes { + guest: Arc, + auth: Option>, + gate: Option>, + attestor: Option>, + scheme: Scheme, +} + +async fn serve_connection( + io: S, + routes: Routes, + // The leaf this connection's handshake actually presented, loaded once at + // accept time. `None` for plaintext. + certificate: Option>, + // Boxed rather than `hyper::Error`: the protocol is chosen per connection + // now, and `auto::Builder` reports failures from whichever of the two it + // ended up speaking. +) -> std::result::Result<(), Box> +where + S: tokio::io::AsyncRead + tokio::io::AsyncWrite + Unpin + Send + 'static, +{ + let Routes { + guest, + auth, + gate, + attestor, + scheme, + } = routes; + let certificate = certificate.map(Arc::new); + let service = hyper::service::service_fn(move |req: hyper::Request| { + let guest = guest.clone(); + let auth = auth.clone(); + let gate = gate.clone(); + let attestor = attestor.clone(); + let certificate = certificate.clone(); + // Cloned per request because the closure is `Fn` — it runs again for + // every request on a kept-alive connection. + let scheme = scheme.clone(); + async move { + // One place, before the router, so the rule covers the guest, + // `/auth/*`, `/enclave/*` and every refusal alike — and so a + // request without a nonce reaches none of them. + // Required whether or not this deployment attests. A client then + // behaves identically either way, and a deployment cannot silently + // stop attesting without its clients noticing — which is the + // failure a client can least afford to miss. + let nonce = match nonce_from_headers(req.headers()) { + Ok(nonce) => nonce, + Err(e) => { + // No document on this one: there is no nonce to bind, and + // one the runtime chose would prove nothing. + tracing::debug!(reason = %e, "refused a request with no usable nonce"); + return Ok::<_, anyhow::Error>(refused( + hyper::StatusCode::BAD_REQUEST, + &e.to_string(), + )); + } + }; + + // Only where a client still has to work out who it is talking to, + // which is the `/auth/` exchange and nothing after it. + // + // A client identifies the enclave on the challenge request — that + // response's document binds the certificate it was served — and + // then pins that certificate for the operation. TLS proves the peer + // holds its private key, which the certificate itself, being + // public, does not; so "the same certificate" and "the same + // enclave" are one statement, and the operation needs no second + // signature to establish what the first already did. + // + // Attesting it anyway cost an NSM signature per operation, and the + // device is the throughput ceiling — `CONCURRENT_DOCUMENTS` is 4. + // That halves the signatures a signed request costs. + let attest_this = req.uri().path().starts_with(crate::auth::AUTH_PREFIX); + + // Generated before the request is routed, not after the + // response exists: it binds the nonce and the connection's + // certificate, neither of which depends on what the guest says. + // Doing it here means a failure refuses before the guest runs, + // rather than stranding a half-produced response. + let document = match &attestor { + Some(attestor) if attest_this => { + match attestor + .document(certificate.as_ref().map(|c| c.as_slice()), &nonce) + .await + { + Ok(document) => Some(document), + Err(e) => { + // 503, not a dropped connection. A bare reset + // is indistinguishable to a client from a + // network fault or an interceptor, and this is + // the one signal that says "this enclave will + // not vouch for itself". + tracing::error!(error = %e, "could not attest a response"); + let mut response = refused( + hyper::StatusCode::SERVICE_UNAVAILABLE, + "this response could not be attested", + ); + response.headers_mut().insert( + hyper::header::CONNECTION, + hyper::header::HeaderValue::from_static("close"), + ); + return Ok(response); + } + } + } + // Either the deployment does not attest, or this is a request + // whose caller has already identified the enclave. + _ => None, + }; + + let mut response = route(req, guest, auth, gate, scheme).await?; + match document { + Some(document) => crate::serve::attest::attach(&mut response, document), + // Not "leave it as it is": a response the runtime did not + // attest must not carry the header at all, or a guest could + // put one there itself. + None => crate::serve::attest::strip(&mut response), + } + Ok(response) + } + }); + + let mut builder = + hyper_util::server::conn::auto::Builder::new(hyper_util::rt::TokioExecutor::new()); + builder.http1().keep_alive(true); + // A stream carrying nothing is indistinguishable from a peer that has gone + // away, and a peer that has gone away is still holding its tenant's slot. + // PING is the only thing at this layer that can tell them apart. + builder + .http2() + .timer(hyper_util::rt::TokioTimer::new()) + .keep_alive_interval(Some(Duration::from_secs(30))) + .keep_alive_timeout(Duration::from_secs(20)); + builder.serve_connection(TokioIo::new(io), service).await +} + +/// Everything that decides what a response *is*, with attestation stripped out. +/// +/// A separate function so the closure above has exactly one exit: that is what +/// makes "no response leaves unattested" and "a guest cannot own the header" +/// single facts rather than five places to audit. +async fn route( + req: hyper::Request, + guest: Arc, + auth: Option>, + gate: Option>, + scheme: Scheme, +) -> Result> { + { + { + let method = req.method().clone(); + let path = req.uri().path().to_string(); + + // `/auth/` is checked first and is not forwardable: a guest able + // to answer there could hand out its own challenges and verify its + // own assertions, which is the same as having none. + if let Some(auth) = &auth { + if path.starts_with(crate::auth::AUTH_PREFIX) { + // `/auth/*` bodies are small and bounded by the endpoints + // themselves; they are read here because the routes take + // bytes rather than a stream. + let body = match http_body_util::BodyExt::collect(req.into_body()).await { + Ok(collected) => collected.to_bytes(), + Err(e) => { + tracing::debug!(error = %e, "reading an auth request body"); + return Ok(refused( + hyper::StatusCode::BAD_REQUEST, + "malformed request", + )); + } + }; + if let Some(response) = auth.handle(&method, &path, &body).await { + return Ok(response); + } + return Ok(refused(hyper::StatusCode::NOT_FOUND, "no such endpoint")); + } + } + + // Everything else is the guest, and **nothing reaches the guest + // without an unspent interaction token.** With no gate configured + // the runtime is unauthenticated by deliberate choice — + // development, and the QEMU harness. + let tenant = match &gate { + Some(gate) => match gate.redeem(&req) { + Ok(verified) => { + // Rebuilt so the token cannot reach the guest. The body + // is forwarded as it arrives, never buffered: an + // interaction may be a stream, and there is no hash to + // hold it still for any more. + let mut rebuilt = hyper::Request::builder() + .method(req.method().clone()) + .uri(req.uri().clone()); + for (name, value) in req.headers() { + if !crate::auth::AUTH_HEADERS + .iter() + .any(|h| name.as_str() == *h) + { + rebuilt = rebuilt.header(name, value); + } + } + use http_body_util::BodyExt; + let body = req + .into_body() + .map_err(wasmtime_wasi_http::p2::bindings::http::types::ErrorCode::from) + .boxed_unsync(); + let rebuilt = rebuilt.body(body).expect("request is well formed"); + return guest + .handle(scheme, rebuilt, Some(&verified.tenant_id)) + .await; + } + Err(denied) => { + // Logged in full, answered in one sentence: which check + // failed is the runtime's business, and telling a + // caller would tell them which guess to refine. + tracing::info!(%path, reason = %denied, "refused an unauthorized interaction"); + return Ok(refused(denied.status(), denied.public_message())); + } + }, + None => None, + }; + guest.handle(scheme, req, tenant.as_ref()).await + } + } +} + +/// The accept loop: TLS if configured, `/enclave/*` to the runtime, the rest +/// to the guest. +/// +/// Nothing about the guest changes between the plaintext and TLS paths, which +/// is why dispatch lives in [`ServeHandle`] and this only decides what wraps +/// it. +pub struct Server { + guest: Arc, + /// Produces the per-response proof. `None` leaves the runtime behaving + /// exactly as it did before attestation existed, nonce included. + attestor: Option>, + /// `/auth/*`, and the gate every other request must pass. + /// + /// Both or neither: routes that issue challenges nobody checks would be + /// worse than no routes at all, and a gate with no way to get a challenge + /// would refuse everything forever. + auth: Option>, + gate: Option>, + tls: Option, + /// Fired once the listener is accepting. ACME waits on it before placing + /// an order, because the challenge is a connection inbound to this socket. + listening: Option>, + addr: SocketAddr, +} + +/// How TLS is terminated. +/// +/// One shape, not one per certificate source. Every connection loads one +/// identity from `slot` — a configuration and the leaf it presents — and keeps +/// it for its life. That is what makes "the certificate this connection is +/// using" answerable: rustls offers no way to ask a live connection, so the +/// pairing is arranged at accept time instead of discovered later. +/// +/// ACME differs only by having a second configuration to offer, so it is a +/// field rather than a variant. As two variants this logic was written twice, +/// which is two places to keep the load-once-and-refuse-resumption discipline +/// and one place to eventually get it wrong. +struct Tls { + /// Where a connection takes its serving identity. Replaced on an ACME + /// renewal — and by the renewal test, which is the point: a connection that + /// already loaded one keeps it. + slot: CertificateSlot, + /// ACME only. The ClientHello chooses: a TLS-ALPN-01 validation connection + /// is answered on this same port, which is the reason that challenge type + /// was chosen. + challenge: Option>, +} + +impl Server { + pub fn new(guest: ServeHandle, addr: SocketAddr) -> Self { + Server { + guest: Arc::new(guest), + attestor: None, + auth: None, + gate: None, + tls: None, + listening: None, + addr, + } + } + + /// Require a verified assertion for everything the guest could see. + pub fn with_authentication(mut self, auth: Arc, gate: Arc) -> Self { + self.auth = Some(auth); + self.gate = Some(gate); + self + } + + pub fn with_attestor(mut self, attestor: Arc) -> Self { + self.attestor = Some(attestor); + self + } + + pub fn with_tls(mut self, certificate: CertificateSlot) -> Self { + self.tls = Some(Tls { + slot: certificate, + challenge: None, + }); + self + } + + pub fn with_acme(mut self, acme: &crate::serve::acme::Acme) -> Self { + self.tls = Some(Tls { + slot: acme.certificate.clone(), + challenge: Some(acme.challenge_config.clone()), + }); + self.listening = Some(acme.listening.clone()); + self + } + + /// Accept connections until the process is stopped. + pub async fn run(self) -> Result<()> { + // Dropping the server cancels the scheduler and its owned worker set. + // A scheduler storage error stops serving instead of losing wakeups. + let mut scheduler = tokio::task::JoinSet::new(); + if let Some(queue) = &self.guest.tasks { + anyhow::ensure!( + self.gate.is_some(), + "background tasks require authentication" + ); + scheduler.spawn(queue.clone().run(self.guest.clone())); + } + // The supervisors, started from what is already on disk. This is what + // re-establishes a connection after a restart — no guest is involved and + // nothing is scheduled; the records are the instruction and this reads + // them. See [`crate::stream`]. + if let Some(streams) = &self.guest.streams { + anyhow::ensure!( + self.gate.is_some(), + "held connections require authentication" + ); + streams.set_egress(self.guest.egress.clone()).await; + scheduler.spawn(streams.clone().run(self.guest.clone())); + } + let listener = TcpListener::bind(self.addr) + .await + .with_context(|| format!("binding {}", self.addr))?; + tracing::info!( + addr = %listener.local_addr()?, + tls = self.tls.is_some(), + attestation = self.attestor.is_some(), + "serving" + ); + + // Only now may ACME order. Until this point a validation connection + // would have found nothing listening, which the CA counts as a failed + // authorization rather than as "try again in a moment". + if let Some(listening) = &self.listening { + listening.notify_one(); + } + + // What the *client* used. With TLS terminated here the guest is still + // told `https`, or it would build wrong absolute URLs and set wrong + // cookie flags. + let scheme = if self.tls.is_some() { + Scheme::Https + } else { + Scheme::Http + }; + let tls = self.tls.map(Arc::new); + + loop { + let (client, peer) = tokio::select! { + accepted = listener.accept() => accepted.context("accepting connection")?, + ended = scheduler.join_next(), if !scheduler.is_empty() => { + ended.context("scheduler disappeared")???; + anyhow::bail!("scheduler unexpectedly stopped"); + } + }; + let routes = Routes { + guest: self.guest.clone(), + auth: self.auth.clone(), + gate: self.gate.clone(), + attestor: self.attestor.clone(), + // `Scheme` is not `Copy`, and the service closure is `Fn` — it + // may run per request on a kept-alive connection. + scheme: scheme.clone(), + }; + let tls = tls.clone(); + + tokio::task::spawn(async move { + // The service closure is built *inside* each arm below, after + // the handshake, because that is the only point at which the + // peer's certificate exists. Built out here — as it was — the + // connection's identity could never reach a request. + let result = match tls.as_deref() { + Some(Tls { slot, challenge }) => { + // The ClientHello is read before any configuration is + // chosen, because under ACME it decides which one: a + // validation connection and the real service share this + // port, which is the reason TLS-ALPN-01 was chosen. + let handshake = + match tokio_rustls::LazyConfigAcceptor::new(Default::default(), client) + .await + { + Ok(handshake) => handshake, + Err(e) => { + tracing::debug!(%peer, error = %e, "TLS handshake failed"); + return; + } + }; + + if let Some(challenge) = challenge { + if rustls_acme::is_tls_alpn_challenge(&handshake.client_hello()) { + // A validation connection carries no HTTP. The + // handshake itself is the proof; completing it + // and closing is the whole exchange. + tracing::info!(%peer, "answering a TLS-ALPN-01 challenge"); + match handshake.into_stream(challenge.clone()).await { + Ok(mut tls) => { + use tokio::io::AsyncWriteExt; + let _ = tls.shutdown().await; + } + Err(e) => { + tracing::warn!( + %peer, error = %e, + "the ACME challenge handshake failed; \ + issuance will not complete" + ); + } + } + return; + } + } + + // Loaded once, and after the challenge check so a + // validation connection is still answered before any + // certificate exists. Everything this connection is + // told about its certificate comes from this value. + let Some(identity) = slot.get() else { + tracing::debug!(%peer, "no certificate to serve with yet"); + return; + }; + match handshake.into_stream(identity.config.clone()).await { + Ok(stream) => { + if resumed(&stream, peer) { + return; + } + serve_connection( + stream, + routes, + Some(identity.certificate_der.clone()), + ) + .await + } + Err(e) => { + // Routine: scanners, health checks, and clients + // that reject the certificate. Under ACME every + // handshake lands here until the first order + // completes, which is the expected state while + // one is in flight. + tracing::debug!(%peer, error = %e, "TLS handshake failed"); + return; + } + } + } + // No TLS, so no certificate and no identity. Whatever the + // client says about itself is discarded, same as any other + // unauthenticated connection. + None => serve_connection(client, routes, None).await, + }; + if let Err(e) = result { + tracing::debug!(%peer, error = %e, "connection ended"); + } + }); + } + } +} + +/// Compile the guest, build the server described by `config`, and serve until +/// the process is stopped. +pub async fn serve_component( + component_bytes: &[u8], + guest: GuestEnvironment, + mut config: ServeConfig, +) -> Result<()> { + // wasmtime 44 enables async at the engine level when the `async` feature + // is on; `Config::async_support` is a no-op. Same as `run_component`. + let engine = ServeHandle::engine_with_watchdog()?; + let mut handle = ServeHandle::new(&engine, component_bytes, guest)? + .with_timeout(config.request_timeout) + .with_max_interaction(config.max_interaction) + .with_egress(config.egress.clone()); + if let Some(tenancy) = &config.tenancy { + handle = handle.with_tenancy(tenancy.clone()); + } + if let Some(limits) = config.background_tasks.take() { + anyhow::ensure!( + config.authentication.is_some(), + "background tasks require authentication" + ); + let queue = crate::tasks::TaskQueue::open( + handle.environment().fs().clone(), + handle.environment().clock().clone(), + limits, + ) + .await?; + handle = handle.with_tasks(queue)?; + } + // Held connections, whenever there is a tenant to attribute one to and a gate to authenticate + // that tenant. No flag of its own: the registry is a directory and a supervisor loop over + // whatever records exist, and a guest that never asks for a connection has neither. What a + // guest may REACH is still the egress allowlist's to say, checked on every dial and every + // send — so enabling this widens nothing. + if config.tenancy.is_some() && config.authentication.is_some() { + let registry = + crate::stream::StreamRegistry::open_registry(handle.environment().fs().clone()).await?; + handle = handle.with_streams(registry)?; + } + let mut notify_forwarder = None; + if let Some(notify) = config.notify.take() { + anyhow::ensure!( + config.authentication.is_some(), + "notifications require authentication: the tenant a wake belongs to comes \ + from a verified assertion, and without a gate there is none" + ); + let registry = + crate::notify::DeviceRegistry::open(handle.environment().fs().clone()).await?; + // Plaintext only where an endpoint override asked for it, which is the + // emulator harness pointing at a local stub. + let plaintext = notify + .endpoint + .as_deref() + .is_some_and(|e| e.starts_with("http://")); + let transport = Arc::new(crate::notify::HttpsTransport::new(plaintext)?); + let mut client = crate::notify::FcmClient::new(notify, transport); + + let clock = handle.environment().clock().clone(); + let now = { + use wasmtime_wasi::HostWallClock; + clock.now().as_millis().min(u64::MAX as u128) as u64 + }; + // Bounded, and never fatal for a transient failure. A push service + // must not be what decides whether the enclave binds its listener — + // but a credential that will never work is a deployment mistake, and + // finding it now beats finding it the first time somebody needed a + // wake signal. + match tokio::time::timeout( + crate::notify::NOTIFY_STARTUP_PROBE_TIMEOUT, + client.probe(now), + ) + .await + { + Ok(Ok(())) => tracing::info!("notifications are configured and the credential works"), + Ok(Err(crate::notify::SendError::Refused(detail))) => { + anyhow::bail!("the FCM credential was rejected: {detail}") + } + Ok(Err(e)) => { + tracing::warn!(error = %e, "could not reach FCM at startup; wake signals will retry") + } + Err(_) => tracing::warn!("FCM did not answer at startup; wake signals will retry"), + } + + let (notifier, forwarder) = crate::notify::start(registry, clock, client); + notify_forwarder = Some(forwarder); + handle = handle.with_notify(notifier)?; + } + + // Before the listener binds. A guest that will not instantiate should stop + // the enclave, not the first client unlucky enough to arrive. + handle.verify_instantiates().await?; + + let mut server = Server::new(handle, config.addr); + match config.authentication { + Some((auth, gate)) => server = server.with_authentication(auth, gate), + None => tracing::warn!( + "serving with NO authentication: every request reaches the guest without a \ + WebAuthn assertion. Set --webauthn-rp-id and --webauthn-origin for anything \ + that is not a development run." + ), + } + + // Where a connection loads its serving identity at accept time. Fixed for a + // self-signed certificate; replaced by the ACME client on issue and on + // every renewal — but a connection that already loaded one keeps it, which + // is what makes "the certificate this connection was served" answerable. + let certificate = match (config.certificate.take(), &config.acme) { + (Some(slot), _) => Some(slot), + (None, Some(acme)) => Some(acme.certificate.clone()), + (None, None) => None, + }; + + match (&certificate, &config.attestation) { + (Some(slot), attestation) => { + if let Some(nsm) = attestation { + // Every response carries a proof, so the device is on the path + // of every request. + let attestor = Arc::new(ResponseAttestor::new(nsm.clone(), component_bytes)); + + // Checked once, here, with the largest nonce a client may send. + // A document too large for a header is a property of this + // deployment's certificate chain, and a runtime that cannot + // attest its responses should refuse to start rather than + // refuse every request. This is also the first call to the + // device, so an NSM that will not answer is found now. + match attestor.verify_fits().await { + Ok(bytes) => { + tracing::info!( + document_header_bytes = bytes, + "attesting every /auth response" + ) + } + Err(e) => anyhow::bail!( + "this runtime cannot attest its responses: {e}. Every request would \ + be refused, so it will not start." + ), + } + server = server.with_attestor(attestor); + } + + // ACME owns its own slot and needs the challenge configuration + // alongside it; a fixed certificate is served straight from the + // slot built above. + server = match &config.acme { + Some(acme) => server.with_acme(acme), + None => server.with_tls(slot.clone()), + }; + } + (None, Some(_)) => { + anyhow::bail!( + "attestation was requested without TLS. A document binds the serving \ + certificate, so without one it would promise a binding it does not have." + ); + } + (None, None) => { + tracing::warn!( + "serving plaintext HTTP with no attestation; \ + inside an enclave this exposes every request to the parent instance" + ); + } + } + + let served = server.run().await; + // Drained after serving, so a wake raised by the last request still has its + // chance to land. Bounded by the forwarder's own flush deadline. + if let Some(forwarder) = notify_forwarder { + forwarder.shutdown().await; + } + served +} + +#[cfg(test)] +mod concurrency_tests { + use super::*; + use tokio::sync::oneshot; + + type HeadResult = std::result::Result< + hyper::Response, + wasmtime_wasi_http::p2::bindings::http::types::ErrorCode, + >; + + fn head() -> HeadResult { + Ok(hyper::Response::new(full(bytes::Bytes::from_static( + b"body", + )))) + } + + /// A guest task, as the dispatch paths build one: it holds `lock` for the + /// whole call, sets a response head part way through, and finishes only + /// when told to. + fn guest_task( + lock: L, + set_head: oneshot::Sender, + finish: oneshot::Receiver<()>, + ) -> tokio::task::JoinHandle> { + tokio::task::spawn(async move { + let _lock = lock; + let _ = set_head.send(head()); + let _ = finish.await; + Ok(()) + }) + } + + /// The property the whole design rests on: a tenant's second request waits + /// for its first to be *finished*, not merely to have produced headers. + #[tokio::test(flavor = "multi_thread")] + async fn one_tenant_serialises_against_itself() { + let tenant = Arc::new(tokio::sync::Mutex::new(())); + let held = tenant.clone().lock_owned().await; + let (set_head, receiver) = oneshot::channel(); + let (finish, finished) = oneshot::channel(); + let task = guest_task(held, set_head, finished); + + // The head is out, and the guest is still writing its body. + await_head( + Duration::from_secs(5), + Duration::from_secs(300), + task, + receiver, + Arc::new(StreamProgress::new()), + ) + .await + .expect("head"); + + let second = tenant.clone().lock_owned(); + assert!( + tokio::time::timeout(Duration::from_millis(100), second) + .await + .is_err(), + "a second request for this tenant started while the first was still running" + ); + + let _ = finish.send(()); + assert!( + tokio::time::timeout(Duration::from_secs(5), tenant.clone().lock_owned()) + .await + .is_ok(), + "the tenant's lock was not released when its guest finished" + ); + } + + /// And the other half: two tenants have nothing to queue on. This is what + /// the removed global semaphore used to prevent. + #[tokio::test(flavor = "multi_thread")] + async fn two_tenants_run_at_the_same_time() { + let alice = Arc::new(tokio::sync::Mutex::new(())); + let bob = Arc::new(tokio::sync::Mutex::new(())); + + let (alice_head, alice_receiver) = oneshot::channel(); + let (alice_finish, alice_finished) = oneshot::channel(); + let alice_task = guest_task(alice.clone().lock_owned().await, alice_head, alice_finished); + await_head( + Duration::from_secs(5), + Duration::from_secs(300), + alice_task, + alice_receiver, + Arc::new(StreamProgress::new()), + ) + .await + .expect("alice's head"); + + // Alice is mid-body. Bob must not be waiting on her. + let (bob_head, bob_receiver) = oneshot::channel(); + let (bob_finish, bob_finished) = oneshot::channel(); + let bob_lock = tokio::time::timeout(Duration::from_millis(100), bob.clone().lock_owned()) + .await + .expect("bob queued behind alice"); + let bob_task = guest_task(bob_lock, bob_head, bob_finished); + let bob_response = tokio::time::timeout( + Duration::from_secs(5), + await_head( + Duration::from_secs(5), + Duration::from_secs(300), + bob_task, + bob_receiver, + Arc::new(StreamProgress::new()), + ), + ) + .await + .expect("bob's head did not arrive while alice was running") + .expect("bob's head"); + assert_eq!(bob_response.status(), 200); + + let _ = alice_finish.send(()); + let _ = bob_finish.send(()); + } + + /// Callers with no resolved tenant are one identity, so they queue behind + /// each other rather than multiplying wasm instances. + #[tokio::test(flavor = "multi_thread")] + async fn anonymous_callers_share_one_slot() { + let anonymous = Arc::new(tokio::sync::Mutex::new(())); + let held = anonymous.clone().lock_owned().await; + let (set_head, receiver) = oneshot::channel(); + let (finish, finished) = oneshot::channel(); + let task = guest_task(held, set_head, finished); + await_head( + Duration::from_secs(5), + Duration::from_secs(300), + task, + receiver, + Arc::new(StreamProgress::new()), + ) + .await + .expect("head"); + + assert!( + tokio::time::timeout(Duration::from_millis(100), anonymous.clone().lock_owned()) + .await + .is_err(), + "a second anonymous request ran alongside the first" + ); + let _ = finish.send(()); + } + + /// A guest that never answers is aborted, and abort drops the task's + /// locals — so an abandoned request must not hold its tenant's lock + /// forever. + #[tokio::test(flavor = "multi_thread")] + async fn an_abandoned_guest_releases_its_tenant() { + let tenant = Arc::new(tokio::sync::Mutex::new(())); + let held = tenant.clone().lock_owned().await; + let (_set_head, receiver) = oneshot::channel::(); + + let task = tokio::task::spawn(async move { + let _lock = held; + std::future::pending::<()>().await; + Ok(()) + }); + + let error = await_head( + Duration::from_millis(50), + Duration::from_secs(300), + task, + receiver, + Arc::new(StreamProgress::new()), + ) + .await + .expect_err("a guest that never answers should be abandoned"); + assert!(format!("{error:#}").contains("abandoned"), "{error:#}"); + + assert!( + tokio::time::timeout(Duration::from_secs(5), tenant.clone().lock_owned()) + .await + .is_ok(), + "an abandoned guest kept its tenant's lock" + ); + } + + /// A guest that fails without setting a response still releases its lock. + #[tokio::test(flavor = "multi_thread")] + async fn a_failed_guest_releases_its_tenant() { + let tenant = Arc::new(tokio::sync::Mutex::new(())); + let held = tenant.clone().lock_owned().await; + let (set_head, receiver) = oneshot::channel::(); + + let task = tokio::task::spawn(async move { + let _lock = held; + drop(set_head); + Err(wasmtime::Error::msg("guest trapped")) + }); + + let error = await_head( + Duration::from_secs(5), + Duration::from_secs(300), + task, + receiver, + Arc::new(StreamProgress::new()), + ) + .await + .expect_err("a trap without a response is an error"); + assert!(format!("{error:#}").contains("guest trapped"), "{error:#}"); + assert!( + tokio::time::timeout(Duration::from_secs(5), tenant.clone().lock_owned()) + .await + .is_ok(), + "a failed guest kept its tenant's lock" + ); + } +} diff --git a/runtime/src/serve/mod.rs b/runtime/src/serve/mod.rs new file mode 100644 index 0000000..5ba91fa --- /dev/null +++ b/runtime/src/serve/mod.rs @@ -0,0 +1,164 @@ +//! Serving a `wasi:http/proxy` guest over plaintext HTTP. +//! +//! The runtime owns the connection, the TLS session and the HTTP parser; the +//! guest receives a parsed request and returns a response. It never sees a +//! socket, a certificate, or a TLS record — [`crate::linker`] does not give it +//! `wasi:sockets` permission to open one and [`EgressPolicy`] refuses the one +//! outbound path `wasi:http` would otherwise offer. +//! +//! That division is the point of terminating TLS here rather than in front of +//! the enclave. A reverse proxy on the parent instance would see every request +//! in the clear, and the parent is precisely the party an enclave exists to +//! exclude. +//! +//! ```text +//! :443 ──TLS──▶ hyper ──▶ /enclave/* ──▶ runtime (attestation, config) +//! └──▶ everything else ──▶ guest +//! ``` + +pub mod acme; +pub mod attest; +pub mod client; +pub mod egress; +mod http; +pub mod pool; +pub mod progress; +pub mod tls; + +pub use acme::{AcmeConfig, CertificateSlot, SealedAcmeCache}; +pub use client::{apply_tenant, X_ENCLAVE_TENANT}; +pub use egress::{EgressAllowlist, Origin}; +pub use http::{serve_component, GuestInstance, ServeConfig, ServeHandle, Server, Tenancy}; +pub use pool::{Checkout, LiveTenant, PoolLimits, Slot, TenantPool}; +pub use tls::{TlsIdentity, TlsMode}; + +use wasmtime_wasi_http::p2::{ + bindings::http::types::ErrorCode, body::HyperOutgoingBody, types::HostFutureIncomingResponse, + types::OutgoingRequestConfig, HttpResult, WasiHttpHooks, +}; + +/// What the guest is allowed to do with `wasi:http/outgoing-handler`. +/// +/// `wasmtime-wasi-http`'s `default-send-request` feature is off in this +/// crate's manifest, which turns `send_request` from a defaulted method into a +/// required one. That is deliberate: with the feature on, a guest importing +/// `outgoing-handler` silently gains real network egress through a rustls +/// instance and root store that nothing in this codebase configures or +/// audits. Making the method mandatory means the answer has to be written +/// down, and this is where it is written down. +/// +/// A guest inside an enclave should not originate connections. Its filesystem +/// is remote already and reached by the *host*, whose S3 traffic is +/// authenticated and encrypted under keys the guest never holds. Egress from +/// the guest itself would be a channel out of the attested boundary carrying +/// whatever the guest chose to put in it. +/// +/// So that is the default, and it stays the default. A deployment whose guest +/// has to reach a service — a wallet cosigner renewing funds with its ASP — +/// names that service's origins, and only those; see [`egress`]. The list is +/// image environment, measured into PCR0, so a client learns where a guest can +/// send traffic from the same attestation that tells it what the guest is. +#[derive(Debug, Clone, Default)] +pub enum EgressPolicy { + /// Every outgoing request fails with `HTTP-request-denied`. + #[default] + Denied, + /// Requests to exactly these origins are sent; every other one is refused + /// as [`EgressPolicy::Denied`] refuses it. + Allowlist(std::sync::Arc), +} + +impl EgressPolicy { + /// `Denied` for an empty list, so an image built with no origins behaves + /// exactly as one that never heard of the setting. + pub fn from_allowlist(allowlist: EgressAllowlist) -> Self { + if allowlist.is_empty() { + EgressPolicy::Denied + } else { + EgressPolicy::Allowlist(std::sync::Arc::new(allowlist)) + } + } +} + +impl WasiHttpHooks for EgressPolicy { + fn send_request( + &mut self, + request: hyper::Request, + config: OutgoingRequestConfig, + ) -> HttpResult { + if let EgressPolicy::Allowlist(list) = self { + if let Some(origin) = list.admits(&request, &config) { + tracing::debug!(%origin, path = %request.uri().path(), "guest egress"); + return Ok(list.send(origin, request, config)); + } + } + // Logged rather than silently refused: a guest attempting egress it was + // not given is either misconfigured or doing something it should not, + // and both are worth seeing in the enclave's console. + tracing::warn!( + uri = %request.uri(), + "guest attempted an outgoing HTTP request; denied by policy" + ); + Err(ErrorCode::HttpRequestDenied.into()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use http_body_util::{BodyExt, Empty}; + + fn empty_request() -> hyper::Request { + hyper::Request::builder() + .uri("https://example.invalid/") + .body( + Empty::::new() + .map_err(|e| match e {}) + .boxed_unsync(), + ) + .unwrap() + } + + /// By default the guest has no outbound network. If this test ever needs + /// changing, the change is a security decision, not a refactor. + #[test] + fn the_guest_cannot_originate_requests() { + let mut policy = EgressPolicy::Denied; + // `let ... else` rather than `expect_err`: the success type is a + // pending response future and does not implement `Debug`. + let Err(err) = policy.send_request(empty_request(), test_config()) else { + panic!("egress must be refused"); + }; + assert!( + format!("{err:?}").contains("HttpRequestDenied"), + "denial must be reported as HTTP-request-denied, got {err:?}" + ); + } + + fn test_config() -> OutgoingRequestConfig { + OutgoingRequestConfig { + use_tls: true, + connect_timeout: std::time::Duration::from_secs(1), + first_byte_timeout: std::time::Duration::from_secs(1), + between_bytes_timeout: std::time::Duration::from_secs(1), + } + } + + /// A deployment's allowlist opens its origins and nothing else: a request + /// anywhere else is refused exactly as with no list at all. + #[test] + fn an_allowlist_refuses_what_it_does_not_name() { + let list = EgressAllowlist::parse(&["https://asp.example.com"]).unwrap(); + let mut policy = EgressPolicy::from_allowlist(list); + let Err(err) = policy.send_request(empty_request(), test_config()) else { + panic!("example.invalid is not on the list"); + }; + assert!(format!("{err:?}").contains("HttpRequestDenied"), "{err:?}"); + } + + #[test] + fn an_empty_allowlist_is_no_egress() { + let policy = EgressPolicy::from_allowlist(EgressAllowlist::parse::<&str>(&[]).unwrap()); + assert!(matches!(policy, EgressPolicy::Denied)); + } +} diff --git a/runtime/src/serve/pool.rs b/runtime/src/serve/pool.rs new file mode 100644 index 0000000..d32ecc3 --- /dev/null +++ b/runtime/src/serve/pool.rs @@ -0,0 +1,342 @@ +//! Live tenants: one filesystem, one warm instance and one lock per client. +//! +//! ## What is kept, and why that and not something else +//! +//! Mounting a client's filesystem derives key material, reads a signed root +//! record, verifies it and opens the object set — measured at 51 µs against a +//! `HashMap`, which in production is that plus two or three S3 round trips. +//! Instantiating the guest is 24 µs and no I/O at all. So **the mount is what +//! is worth keeping warm**; the instance rides along because it is nearly free +//! and it is convenient to rebuild the two together. +//! +//! ## The lock is a correctness mechanism, not only a cache +//! +//! Each client has its own `tokio::Mutex`, and that is the whole concurrency +//! model: **a client is serialised against itself and against nobody else.** +//! +//! Serialised against itself, because a cosigner reserving a nonce must not +//! race its own second request. Not against anyone else, because a client's +//! filesystem is theirs alone — separate `Store`, separate transaction lock — +//! so there is nothing for two clients to contend over, and making them queue +//! would mean one client's slow commit stalling everybody. +//! +//! ## Why anonymous callers get nothing +//! +//! A slot is keyed by the SHA-256 of a client's TLS public key, which the +//! handshake proves. A caller who presented no certificate has no key to be +//! identified by, so they get a fresh instance over the runtime's own +//! filesystem and never occupy a slot — otherwise anyone who could open a +//! socket could fill the pool and evict every real client's mount. + +use std::collections::HashMap; +use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; +use std::sync::{Arc, Mutex}; +use std::time::{Duration, Instant}; + +use s3fs_core::Inode; + +/// How large the pool may grow, and how long an idle tenant is kept. +/// +/// Both matter more than they look. A tenant costs a `BlockStore` cache — +/// `block_cache_bytes`, 64 MiB by default — plus a wasm linear memory, so a +/// pool sized by hope rather than arithmetic is an enclave that dies of memory +/// exhaustion under exactly the load it was built for. +#[derive(Debug, Clone)] +pub struct PoolLimits { + pub max_tenants: usize, + pub idle_timeout: Duration, + /// Calls one instance serves before it is rebuilt. Wasm linear memory + /// never shrinks, so an instance that lives forever only grows. + pub max_requests_per_instance: u64, +} + +impl Default for PoolLimits { + fn default() -> Self { + PoolLimits { + max_tenants: 64, + idle_timeout: Duration::from_secs(900), + max_requests_per_instance: 10_000, + } + } +} + +/// A client's view of the filesystem, and its warm instance. +/// +/// `scope` is the directory this client's guest calls `/`. The filesystem +/// itself is shared and is not here — one mount, one block cache, one +/// transaction stream for every tenant. +/// +/// `instance` is an `Option` so a trap can discard the poisoned linear memory +/// while keeping the resolved directory. That is cheap either way; what it +/// mainly buys is that one client's trap is visible to nobody else. +pub struct LiveTenant { + pub scope: Arc, + pub tenant_id: [u8; 16], + pub instance: Option, + pub requests: u64, +} + +/// One client's slot. +/// +/// `last_used` and `in_flight` sit *outside* the async mutex on purpose: the +/// eviction sweep has to compare tenants without awaiting on any of them, and a +/// sweep that could block on a busy tenant would be a sweep that stalls the +/// server it is tidying. +pub struct Slot { + /// `Arc`'d so a request can take an *owned* guard and carry it into the + /// task that runs the guest — the lock has to outlive the response head, + /// and a borrowed guard could not. + tenant: Arc>>>, + last_used: AtomicU64, + in_flight: AtomicUsize, +} + +impl Slot { + fn new(now: u64) -> Self { + Slot { + tenant: Arc::new(tokio::sync::Mutex::new(None)), + last_used: AtomicU64::new(now), + in_flight: AtomicUsize::new(0), + } + } + + pub fn tenant(&self) -> &Arc>>> { + &self.tenant + } +} + +/// A slot checked out for one request. +/// +/// Holding this is what marks the tenant busy, so the sweep leaves it alone; +/// dropping it releases that mark however the request ended, including a panic. +pub struct Checkout { + slot: Arc>, +} + +impl Checkout { + pub fn slot(&self) -> &Arc> { + &self.slot + } +} + +impl Drop for Checkout { + fn drop(&mut self) { + self.slot.in_flight.fetch_sub(1, Ordering::Release); + } +} + +pub struct TenantPool { + /// A `std::sync::Mutex`, not a `tokio` one, and never held across an + /// await: everything under it is a hash lookup and an `Arc` clone. The + /// *per-tenant* lock is the async one, and it is a different lock for a + /// different job. + slots: Mutex>>>, + limits: PoolLimits, + started: Instant, +} + +impl TenantPool { + pub fn new(limits: PoolLimits) -> Self { + TenantPool { + slots: Mutex::new(HashMap::new()), + limits: limits.clone(), + started: Instant::now(), + } + } + + pub fn limits(&self) -> &PoolLimits { + &self.limits + } + + fn now(&self) -> u64 { + self.started.elapsed().as_millis() as u64 + } + + /// Check out this client's slot, creating an empty one if needed. + /// + /// The slot, not the tenant: the caller then takes the async lock and + /// builds the filesystem under it if it is not there yet. That ordering is + /// what makes two simultaneous first requests from one client produce + /// **one** filesystem — they find the same slot, and the second waits on + /// the lock the first is holding rather than racing it into a second mount. + pub fn checkout(&self, tenant: &[u8; 16]) -> Checkout { + let now = self.now(); + let mut slots = self.slots.lock().expect("tenant pool mutex poisoned"); + + if let Some(slot) = slots.get(tenant) { + slot.last_used.store(now, Ordering::Release); + slot.in_flight.fetch_add(1, Ordering::Release); + return Checkout { slot: slot.clone() }; + } + + // Room first, so the pool never exceeds its cap even briefly. + self.evict_while_full(&mut slots, now); + + let slot = Arc::new(Slot::new(now)); + slot.in_flight.fetch_add(1, Ordering::Release); + slots.insert(*tenant, slot.clone()); + Checkout { slot } + } + + /// Drop idle and over-cap tenants. + /// + /// Removing a slot from the map does not destroy it: a request already + /// holding the `Arc` keeps working and the filesystem unmounts when it + /// finishes. So eviction can never pull a filesystem out from under a call + /// in progress — it only stops future requests finding it. + fn evict_while_full(&self, slots: &mut HashMap<[u8; 16], Arc>>, now: u64) { + let idle_ms = self.limits.idle_timeout.as_millis() as u64; + slots.retain(|_, slot| { + slot.in_flight.load(Ordering::Acquire) > 0 + || now.saturating_sub(slot.last_used.load(Ordering::Acquire)) < idle_ms + }); + + while slots.len() >= self.limits.max_tenants.max(1) { + let victim = slots + .iter() + .filter(|(_, s)| s.in_flight.load(Ordering::Acquire) == 0) + .min_by_key(|(_, s)| s.last_used.load(Ordering::Acquire)) + .map(|(k, _)| *k); + match victim { + Some(key) => { + slots.remove(&key); + } + // Every tenant is busy. Refusing to evict is right: the + // alternative is unmounting a filesystem someone is mid-commit + // on. The pool runs slightly over its cap until they finish, + // which the global concurrency limit already bounds. + None => break, + } + } + } + + /// Tenants currently held. Diagnostics, and the assertion tests need. + pub fn len(&self) -> usize { + self.slots.lock().expect("tenant pool mutex poisoned").len() + } + + pub fn is_empty(&self) -> bool { + self.len() == 0 + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// A stand-in for the wasm instance: the pool never looks inside one. + #[derive(Debug, PartialEq, Eq)] + struct FakeInstance(u32); + + fn pool(max: usize) -> TenantPool { + TenantPool::new(PoolLimits { + max_tenants: max, + ..Default::default() + }) + } + + #[tokio::test] + async fn a_client_returns_to_the_same_slot() { + let pool = pool(8); + let client = [0xab; 16]; + + { + let first = pool.checkout(&client); + *first.slot().tenant().lock().await = None; + } + let again = pool.checkout(&client); + assert_eq!(pool.len(), 1, "a second request made a second slot"); + drop(again); + } + + #[tokio::test] + async fn two_clients_get_two_slots() { + let pool = pool(8); + let a = pool.checkout(&[0xaa; 16]); + let b = pool.checkout(&[0xbb; 16]); + assert_eq!(pool.len(), 2); + assert!( + !Arc::ptr_eq(a.slot(), b.slot()), + "two clients shared a slot" + ); + } + + /// The property the whole design is for: one client's request does not + /// wait on another's. If these locks were shared this would deadlock. + #[tokio::test] + async fn one_client_does_not_block_another() { + let pool = pool(8); + let a = pool.checkout(&[0xaa; 16]); + let b = pool.checkout(&[0xbb; 16]); + + let held = a.slot().tenant().lock().await; + // B proceeds while A's lock is held, and needs no timeout to do it. + let other = b.slot().tenant().lock().await; + drop(other); + drop(held); + } + + /// And the other half: a client *is* serialised against itself, which is + /// what makes a nonce reservation safe. + #[tokio::test] + async fn a_client_is_serialised_against_itself() { + let pool = pool(8); + let client = [0xab; 16]; + let first = pool.checkout(&client); + let second = pool.checkout(&client); + assert!(Arc::ptr_eq(first.slot(), second.slot())); + + let held = first.slot().tenant().lock().await; + assert!( + second.slot().tenant().try_lock().is_err(), + "two requests from one client held the tenant at once" + ); + drop(held); + } + + #[tokio::test] + async fn the_pool_stays_within_its_cap() { + let pool = pool(4); + for i in 0..32u8 { + let mut client = [0u8; 16]; + client[0] = i; + drop(pool.checkout(&client)); + } + assert!(pool.len() <= 4, "pool grew to {}", pool.len()); + } + + /// Eviction must never unmount a filesystem a request is using. A busy + /// tenant is skipped even when the pool is over its cap. + #[tokio::test] + async fn a_busy_tenant_is_never_evicted() { + let pool = pool(2); + let busy = pool.checkout(&[0xff; 16]); + + for i in 0..16u8 { + let mut client = [0u8; 16]; + client[0] = i; + drop(pool.checkout(&client)); + } + + let again = pool.checkout(&[0xff; 16]); + assert!( + Arc::ptr_eq(busy.slot(), again.slot()), + "a slot in use was evicted and rebuilt" + ); + } + + /// Dropping a checkout releases the tenant however the request ended. + #[tokio::test] + async fn finishing_a_request_makes_a_tenant_evictable_again() { + let pool = pool(2); + let client = [0xff; 16]; + drop(pool.checkout(&client)); + + for i in 0..8u8 { + let mut other = [0u8; 16]; + other[0] = i; + drop(pool.checkout(&other)); + } + assert!(pool.len() <= 2); + } +} diff --git a/runtime/src/serve/progress.rs b/runtime/src/serve/progress.rs new file mode 100644 index 0000000..4b05c5b --- /dev/null +++ b/runtime/src/serve/progress.rs @@ -0,0 +1,236 @@ +//! Evidence that a call is still doing something. +//! +//! The epoch watchdog gives a guest a fixed budget of wall clock and traps it +//! when the budget runs out. That is right for a request/response call, where +//! taking longer than the timeout means something is wrong. It is wrong for a +//! long-lived stream, where taking a long time is the *point* — a signing +//! session may be open for minutes and be healthy throughout. +//! +//! Two things must stay true at once: a stream doing real work lives, and a +//! guest spinning in a loop still dies. The distinguishing fact is not elapsed +//! time but whether bytes moved, so that is what this measures. +//! +//! # Why a guest cannot fake it +//! +//! Neither counter is under the guest's control: +//! +//! - the **request** counter is fed by bytes hyper delivered from the client, +//! so it advances only when the peer actually sends; +//! - the **response** counter is fed by bytes hyper pulled *out* of the +//! outgoing body, so it advances only when the client actually reads. A guest +//! writing into a buffer nobody drains moves it exactly once, because the +//! channel behind it holds two chunks and then stops being polled. +//! +//! So "made progress" means the guest and its peer are still talking, which is +//! the only thing worth keeping a tenant's slot open for. + +use std::pin::Pin; +use std::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, Ordering}; +use std::task::{Context, Poll}; + +use bytes::Bytes; +use hyper::body::{Body, Frame}; + +/// Bytes moved on one call, and what the watchdog saw last time it looked. +#[derive(Debug, Default)] +pub struct StreamProgress { + /// Both directions together. They are never reported apart, and a stream + /// that is only receiving is as alive as one that is only sending. + bytes: AtomicU64, + /// Written only by [`StreamProgress::still_working`]. Not a count of + /// anything — it exists to be compared with `bytes` one epoch later. + last_seen: AtomicU64, + /// Whether the response head has gone out. + /// + /// This is what decides how patient the watchdog may be, and the reason is + /// a race the module doc on `watchdog` spells out. Until the head is set, + /// `await_head` is holding a `tokio::time::timeout` that will `abort` the + /// task — and abort needs an await point, which a spinning guest never + /// reaches. The epoch has to win that race, so before the head there is no + /// patience at all: the first silent expiry ends the call, exactly as it + /// did before any of this existed. + /// + /// Once the head is out that timeout is gone — nothing will abort the call + /// — so silence can be judged properly instead of instantly. + head_sent: AtomicBool, + /// Consecutive expiries that saw nothing move. + /// + /// One silent observation is not evidence of a runaway. The response + /// counter advances when hyper *drains* the body, which is a different + /// moment from the guest's epoch check, so a busy machine can leave a + /// perfectly healthy stream looking idle for one tick. Requiring several + /// in a row is what separates "the reader is behind" from "nothing is + /// happening", and it costs a runaway only the strikes it takes to die. + strikes: AtomicU32, +} + +impl StreamProgress { + pub fn new() -> Self { + Self::default() + } + + /// Count bytes that crossed the boundary in either direction. + pub fn record(&self, bytes: usize) { + self.bytes.fetch_add(bytes as u64, Ordering::Relaxed); + } + + /// Start a new call on a store that may have served others. + /// + /// A pooled instance keeps its counters between requests, and a fresh call + /// must not inherit the last one's progress — it would be handed a free + /// extension it did not earn. + pub fn reset(&self) { + self.bytes.store(0, Ordering::Relaxed); + self.last_seen.store(0, Ordering::Relaxed); + self.strikes.store(0, Ordering::Relaxed); + self.head_sent.store(false, Ordering::Relaxed); + } + + /// The guest has set a response; `await_head` is no longer watching. + pub fn head_sent(&self) { + self.head_sent.store(true, Ordering::Relaxed); + } + + /// Should this call be allowed to keep running? + /// + /// **Not idempotent.** Asking *is* the observation: it records what it saw + /// so the next call compares against this moment, and it counts the strike. + /// Only the epoch callback may call it, and only once per expiry. + /// + /// Traffic clears the record. `limit` consecutive silent expiries is what + /// it takes to be judged a runaway, which bounds how long one can burn a + /// worker after its budget ran out — independent of how generous that + /// budget was. + pub fn still_working(&self, limit: u32) -> bool { + let now = self.bytes.load(Ordering::Relaxed); + if self.last_seen.swap(now, Ordering::Relaxed) != now { + self.strikes.store(0, Ordering::Relaxed); + return true; + } + // Before the head, the epoch is the only thing that can stop this + // guest and it must do so now — see `head_sent`. + if !self.head_sent.load(Ordering::Relaxed) { + return false; + } + self.strikes.fetch_add(1, Ordering::Relaxed) + 1 < limit + } +} + +/// A body that reports what passes through it and changes nothing else. +/// +/// Deliberately not a place for policy: it does not cap, delay or inspect. The +/// runtime holds both the request body before it becomes a guest resource and +/// the response body after it stops being one, so wrapping at those two points +/// observes the whole call without a fork of `wasmtime-wasi-http`. +pub struct Counting { + inner: B, + progress: std::sync::Arc, +} + +impl Counting { + pub fn new(inner: B, progress: std::sync::Arc) -> Self { + Counting { inner, progress } + } +} + +impl Body for Counting +where + B: Body + Unpin, +{ + type Data = Bytes; + type Error = B::Error; + + fn poll_frame( + self: Pin<&mut Self>, + cx: &mut Context<'_>, + ) -> Poll, Self::Error>>> { + let this = self.get_mut(); + let polled = Pin::new(&mut this.inner).poll_frame(cx); + if let Poll::Ready(Some(Ok(frame))) = &polled { + if let Some(data) = frame.data_ref() { + this.progress.record(data.len()); + } + } + polled + } + + fn is_end_stream(&self) -> bool { + self.inner.is_end_stream() + } + + fn size_hint(&self) -> hyper::body::SizeHint { + self.inner.size_hint() + } +} + +#[cfg(test)] +mod tests { + use super::*; + use http_body_util::{BodyExt, Full}; + + #[tokio::test] + async fn a_counting_body_reports_exactly_what_passed_through() { + let progress = std::sync::Arc::new(StreamProgress::new()); + let body = Counting::new( + Full::new(Bytes::from_static(b"twelve bytes")), + progress.clone(), + ); + let collected = body.collect().await.expect("collecting").to_bytes(); + assert_eq!(collected.len(), 12); + assert!( + progress.still_working(2), + "twelve bytes moved and went unreported" + ); + } + + /// The property the watchdog rests on: silence reads as silence, but only + /// once it has been silent long enough to mean something. + #[test] + fn sustained_silence_is_what_ends_a_call() { + let progress = StreamProgress::new(); + progress.head_sent(); + // One quiet tick is tolerated: the reader may simply be behind. + assert!(progress.still_working(3), "one quiet tick ended the call"); + assert!(progress.still_working(3), "two quiet ticks ended the call"); + assert!(!progress.still_working(3), "silence never ended the call"); + } + + /// Traffic clears the record, so a stream that pauses and resumes is not + /// carrying strikes from an earlier lull. + #[test] + fn moving_bytes_forgives_earlier_silence() { + let progress = StreamProgress::new(); + progress.head_sent(); + assert!(progress.still_working(2), "the first quiet tick"); + progress.record(1); + assert!( + progress.still_working(2), + "traffic did not clear the strike" + ); + assert!(progress.still_working(2), "the strike count was not reset"); + } + + /// Before the head there is no patience: the epoch must beat `await_head`. + #[test] + fn a_guest_that_has_not_answered_yet_gets_no_grace() { + let progress = StreamProgress::new(); + assert!( + !progress.still_working(8), + "a silent guest was given grace it could not be aborted out of" + ); + } + + /// A pooled instance must not hand its next request a free extension. + #[test] + fn a_reset_forgets_what_the_last_call_moved() { + let progress = StreamProgress::new(); + progress.record(4096); + progress.reset(); + // A limit of one makes the very first silent observation decisive, + // which is the sharpest way to ask "did anything carry over?". + assert!( + !progress.still_working(1), + "the previous call's bytes counted for this one" + ); + } +} diff --git a/runtime/src/serve/tls.rs b/runtime/src/serve/tls.rs new file mode 100644 index 0000000..567e669 --- /dev/null +++ b/runtime/src/serve/tls.rs @@ -0,0 +1,266 @@ +//! The TLS certificate the enclave serves, and where its key comes from. +//! +//! The key is generated **inside** the enclave and never leaves it. That is +//! the property the whole design rests on: a certificate whose private key +//! only ever existed in attested memory, whose hash the NSM then signs into an +//! attestation document. A client that checks the binding knows its TLS +//! session terminates in that enclave — not in the EC2 instance hosting it, +//! and not in a proxy the operator controls. +//! +//! Terminating TLS on the parent instance and forwarding plaintext would be +//! far simpler and would give away the entire point. +//! +//! ## Where the key's randomness comes from +//! +//! `rcgen` generates through `aws-lc-rs`, which draws from the kernel. That +//! deserves a note, because [`crate::random`] goes to some trouble to keep the +//! *guest's* `wasi:random` off the kernel pool and on the NSM directly. +//! +//! The two cases differ. The guest's concern is that the runtime might not be +//! in an enclave at all, and nothing in the kernel pool would say so — a +//! silent downgrade. Here, the enclave kernel has exactly one entropy source, +//! the NSM, which it uses to seed its pool before userspace starts; the boot +//! log shows `NSM RNG: returning rand bytes` immediately followed by +//! `random: crng init done`. So inside an enclave the kernel pool *is* NSM +//! entropy. +//! +//! What makes that argument safe to rely on is that it is checked rather than +//! assumed: an enclave image sets `S3FS_RANDOM_SOURCE=nsm`, which refuses to +//! start without a working `/dev/nsm`. If the runtime got that far, it is in +//! an enclave. + +use std::sync::Arc; + +use anyhow::{Context, Result}; +use rustls::pki_types::{CertificateDer, PrivateKeyDer}; + +/// A serving certificate and the rustls configuration built from it. +pub struct TlsIdentity { + /// DER of the leaf certificate — the bytes a client hashes to check the + /// attestation binding, and therefore what goes into `user_data`. + pub certificate_der: Vec, + pub config: Arc, +} + +impl std::fmt::Debug for TlsIdentity { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("TlsIdentity") + .field( + "certificate_sha256", + &hex::encode(nitro_attestation::sha256(&self.certificate_der)), + ) + .finish() + } +} + +impl TlsIdentity { + /// Build from a certificate chain and key already in hand — the ACME path. + /// + /// `chain` is leaf first, as rustls and every ACME server present it. + pub fn from_chain(chain: Vec>, key: PrivateKeyDer<'static>) -> Result { + let leaf = chain.first().context("certificate chain is empty")?.clone(); + + let certs: Vec> = + chain.into_iter().map(CertificateDer::from).collect(); + // The provider is named rather than left to `ServerConfig::builder()`, + // which resolves it from rustls's compiled-in features and **panics** + // when more than one is present. That is not hypothetical here: the + // AWS SDK brings rustls with `ring` while this crate asks for + // `aws-lc-rs`, so in any build with the `aws` feature both exist and + // there is no unambiguous default. Naming it also keeps the whole + // image on one implementation of these primitives. + let config = rustls::ServerConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .context("selecting TLS protocol versions")? + // No client certificates. Authentication is a WebAuthn assertion bound + // to one request — see `crate::auth` — and a certificate would be a + // second, weaker way to become a tenant that could not bind an + // approval to a transaction. + .with_no_client_auth() + .with_single_cert(certs, key) + .context("building the TLS configuration")?; + + // No session resumption. This is what makes "the certificate this + // connection is using" a well-defined thing to attest. + // + // rustls calls the certificate resolver while processing every + // ClientHello, but sends a Certificate message **only on a full + // handshake** (`server/tls13.rs`, gated on `full_handshake`). A resumed + // session is authenticated by whatever the client cached from its + // original handshake — so after an ACME renewal a resumed connection is + // running on the old certificate while the server has the new one in + // hand. Attesting the current certificate there would name one the + // client's session was never authenticated under: a false binding, in + // exactly the case per-response attestation exists to get right. + // + // rustls exposes no way to ask which certificate an earlier session + // used, so the only correct answer is to have no earlier session. The + // cost is one handshake signature per connection, which is nothing + // beside the per-response attestation this enables. + let mut config = config; + config.session_storage = Arc::new(rustls::server::NoServerSessionStorage {}); + config.send_tls13_tickets = 0; + + // Preference order, and both are offered because both are served. A + // gRPC client offers only `h2` and must be given it; curl and a browser + // offer `http/1.1`, or no ALPN at all, and rustls skips the extension + // entirely for the latter. The one client this turns away is one that + // offers ALPN with no overlap, which is correct: there is no protocol + // in common to speak. + // + // Unrelated to the TLS-ALPN-01 challenge, which never reaches this + // configuration — `rustls_acme` builds its own advertising + // `acme-tls/1`, chosen by inspecting the ClientHello before any + // serving identity is consulted. + config.alpn_protocols = vec![b"h2".to_vec(), b"http/1.1".to_vec()]; + + Ok(TlsIdentity { + certificate_der: leaf, + config: Arc::new(config), + }) + } + + /// Generate a self-signed certificate for `domains`. + /// + /// **Not available in a production build.** The runtime serves ACME-issued + /// certificates and nothing else: a certificate the runtime minted for + /// itself is one an operator can mint too, so it cannot distinguish this + /// enclave from a process impersonating it — the attestation binding is + /// what carries that weight, and it binds whatever certificate is being + /// served, including one that was forged. + /// + /// It survives behind `testing` because the test harnesses and the QEMU + /// emulator have no domain and no reachable CA. A binary built without that + /// feature refuses `--tls self-signed` outright rather than quietly + /// generating one, and `rcgen` is not in it at all. + #[cfg(any(test, feature = "testing"))] + pub fn self_signed(domains: &[String]) -> Result { + let names: Vec = if domains.is_empty() { + vec!["localhost".to_string()] + } else { + domains.to_vec() + }; + + let key = rcgen::KeyPair::generate_for(&rcgen::PKCS_ECDSA_P384_SHA384) + .context("generating the TLS key")?; + let certificate = rcgen::CertificateParams::new(names.clone()) + .context("building certificate parameters")? + .self_signed(&key) + .context("self-signing the certificate")?; + + let key_der = PrivateKeyDer::try_from(key.serialize_der()) + .map_err(|e| anyhow::anyhow!("encoding the TLS key: {e}"))?; + let identity = Self::from_chain(vec![certificate.der().to_vec()], key_der)?; + + tracing::info!( + domains = ?names, + certificate_sha256 = %hex::encode(nitro_attestation::sha256(&identity.certificate_der)), + "generated a self-signed TLS certificate inside the enclave" + ); + Ok(identity) + } +} + +/// Where the serving certificate comes from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum TlsMode { + /// Plaintext HTTP. Development, and behind a trusted terminator only — + /// which inside an enclave means nowhere. + Off, + /// Obtain one from an ACME provider over TLS-ALPN-01. + Acme, +} + +impl TlsMode { + pub fn parse(s: &str) -> Result { + match s.trim().to_ascii_lowercase().as_str() { + "off" | "none" | "plaintext" => Ok(TlsMode::Off), + // Named explicitly so a deployment that asks for it is told why it + // cannot have it, rather than "expected one of …". No build has + // this mode, testing included: the QEMU harness was the last user + // and it now runs a real ACME order against a local Pebble, which + // is the path production takes. + "self-signed" | "selfsigned" => Err( + "this runtime serves ACME-issued certificates only. A self-signed \ + certificate is one an operator can mint too, so it cannot tell this \ + enclave apart from something impersonating it. Use --tls acme, \ + pointing --acme-directory at a test CA if there is no public one." + .to_string(), + ), + "acme" | "letsencrypt" => Ok(TlsMode::Acme), + other => Err(format!("expected one of off, acme; got {other:?}")), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_self_signed_identity_has_a_usable_certificate() { + let identity = TlsIdentity::self_signed(&["enclave.test".to_string()]).unwrap(); + assert!(!identity.certificate_der.is_empty()); + + // It must parse as X.509, or the hash going into the attestation + // document would bind bytes no client could reproduce. + use x509_parser::prelude::FromDer; + let (_, cert) = + x509_parser::certificate::X509Certificate::from_der(&identity.certificate_der).unwrap(); + assert!(cert + .subject_alternative_name() + .unwrap() + .is_some_and(|san| format!("{:?}", san.value).contains("enclave.test"))); + } + + /// Two enclaves, two keys. A shared certificate would let one enclave's + /// attestation vouch for another's connections. + #[test] + fn each_identity_gets_its_own_key() { + let a = TlsIdentity::self_signed(&[]).unwrap(); + let b = TlsIdentity::self_signed(&[]).unwrap(); + assert_ne!(a.certificate_der, b.certificate_der); + } + + #[test] + fn modes_parse_the_documented_values() { + assert_eq!(TlsMode::parse("off"), Ok(TlsMode::Off)); + assert_eq!(TlsMode::parse("ACME"), Ok(TlsMode::Acme)); + assert!(TlsMode::parse("maybe").is_err()); + } + + /// No build serves a certificate it minted itself, and the refusal says + /// what to do instead — a runtime that answered "expected one of off, acme" + /// would leave an operator guessing at a decision that was deliberate. + #[test] + fn a_self_signed_deployment_is_refused_with_a_reason() { + let err = TlsMode::parse("self-signed").unwrap_err(); + assert!(err.contains("ACME-issued certificates only"), "{err}"); + assert!(err.contains("--acme-directory"), "{err}"); + } + + /// Both protocols, in preference order, on every serving identity. + /// + /// `from_chain` is the single constructor — `self_signed` and the ACME path + /// both route through it — so this is the one place the list is stated and + /// there is nowhere for two identities to disagree about what they speak. + #[test] + fn a_serving_identity_offers_h2_and_http11_in_that_order() { + let identity = TlsIdentity::self_signed(&["enclave.test".to_string()]) + .expect("a self-signed identity"); + assert_eq!( + identity.config.alpn_protocols, + vec![b"h2".to_vec(), b"http/1.1".to_vec()], + "gRPC needs h2 offered, and everything else needs http/1.1 kept" + ); + } + + #[test] + fn an_empty_chain_is_refused() { + let key = rcgen::KeyPair::generate_for(&rcgen::PKCS_ECDSA_P384_SHA384).unwrap(); + let key_der = PrivateKeyDer::try_from(key.serialize_der()).unwrap(); + assert!(TlsIdentity::from_chain(vec![], key_der).is_err()); + } +} diff --git a/runtime/src/state.rs b/runtime/src/state.rs new file mode 100644 index 0000000..3c15e09 --- /dev/null +++ b/runtime/src/state.rs @@ -0,0 +1,242 @@ +//! The store-data type both binaries hand to wasmtime. + +use std::sync::Arc; + +use crate::wasi::descriptors::Descriptor; +use crate::wasi::{S3FsCtxView, S3WasiView}; +use s3fs_core::{Fs, Inode}; +use wasmtime::component::ResourceTable; +use wasmtime::{Result, Store}; +use wasmtime_wasi::{WasiCtx, WasiCtxView, WasiView}; + +/// Implements both `WasiView` (so `wasmtime-wasi` can serve `wasi:io`, +/// `wasi:cli`, clocks, random, sockets) and `S3WasiView` (so [`crate::wasi`] +/// can serve `wasi:filesystem`). +pub struct State { + pub(crate) tasks: Option, + /// The connections the runtime holds for this tenant — see [`crate::stream`]. + pub(crate) streams: Option, + pub(crate) notify: Option, + wasi: WasiCtx, + table: ResourceTable, + fs: Arc, + /// What this guest sees as `/` — see [`S3FsCtxView::scope`]. + scope: Arc, + /// The same clock the guest sees through `wasi:clocks`, so `set-times` + /// with "now" agrees with it. + clock: Arc, + http: wasmtime_wasi_http::WasiHttpCtx, + /// What `wasi:http/outgoing-handler` may reach. `Denied` until the + /// serving code says otherwise — see [`State::set_egress`]. + egress: crate::serve::EgressPolicy, +} + +impl State { + pub fn new(wasi: WasiCtx, fs: Arc, clock: Arc) -> Self { + let scope = fs.root(); + Self::scoped(wasi, fs, scope, clock) + } + + /// A `State` whose guest sees `scope` as `/` and can name nothing above it. + pub fn scoped( + wasi: WasiCtx, + fs: Arc, + scope: Arc, + clock: Arc, + ) -> Self { + State { + tasks: None, + streams: None, + notify: None, + wasi, + scope, + table: ResourceTable::new(), + fs, + clock, + http: wasmtime_wasi_http::WasiHttpCtx::new(), + egress: crate::serve::EgressPolicy::Denied, + } + } + + pub fn fs(&self) -> &Arc { + &self.fs + } + + /// Give the guest the deployment's egress policy. + pub fn set_egress(&mut self, egress: crate::serve::EgressPolicy) { + self.egress = egress; + } + + /// Whether the guest left anything behind in the resource table. + /// + /// For a `State` that lives one request the table goes with it, and the + /// `Drop` below releases whatever files it still held. It matters for a + /// *pooled* instance: the host pushes an `incoming-request` and a + /// `response-outparam` per call and never removes them, so everything + /// here is reclaimed by the guest dropping its handles. A guest that does + /// not is not leaking unboundedly — the table is a slab with a free list + /// — but it is leaving entries a later request from the same client could + /// still address. + pub fn resources_settled(&self) -> bool { + self.table.is_empty() + } +} + +/// A file the guest opened and never dropped is held open by the filesystem's +/// own handle table, not only by this one: dropping the resource table on a +/// trap, an abort or an eviction releases the guest's reference and nothing +/// else. Every discard path ends here, so this is the one place to let go. +/// +/// Abandoned, not closed: a close flushes, and `Drop` cannot wait for one. A +/// flush started here would land whenever it was scheduled — after the +/// tenant's lock had passed to its next request, and possibly on top of what +/// that request committed to the same file. So a guest that did not finish +/// loses what it had not synced, as it would on a crash. Abandoning can still +/// free an unlinked file, which is async, hence the spawn; without a runtime +/// there is nothing to spawn on, and the handles stay open until the process +/// does, which is what would have happened anyway. +impl Drop for State { + fn drop(&mut self) { + let handles: Vec<_> = self + .table + .iter_mut() + .filter_map(|entry| match entry.downcast_ref::() { + Some(Descriptor::File { handle, .. }) => Some(handle.clone()), + _ => None, + }) + .collect(); + if handles.is_empty() { + return; + } + let Ok(rt) = tokio::runtime::Handle::try_current() else { + tracing::warn!( + open = handles.len(), + "guest files left open with no runtime to close them" + ); + return; + }; + let fs = self.fs.clone(); + rt.spawn(async move { + for h in handles { + if let Err(e) = fs.abandon(&h).await { + tracing::warn!(error = %e, "releasing a file a discarded guest left open"); + } + } + }); + } +} + +impl WasiView for State { + fn ctx(&mut self) -> WasiCtxView<'_> { + WasiCtxView { + ctx: &mut self.wasi, + table: &mut self.table, + } + } +} + +impl S3WasiView for State { + fn s3fs_view(&mut self) -> S3FsCtxView<'_> { + // The SAME `ResourceTable` instance `wasmtime-wasi` uses. Stream + // resources this crate pushes are resolved by wasmtime-wasi's stream + // methods, so two tables would produce handles that look valid and + // resolve to nothing. + S3FsCtxView { + fs: &self.fs, + scope: &self.scope, + table: &mut self.table, + clock: &self.clock, + } + } +} + +/// `wasi:http` needs its own context and its own hooks, projected out of the +/// same `ResourceTable` as everything else — a second table would hand the +/// guest stream handles that look valid and resolve to nothing, the same trap +/// documented on [`S3WasiView`]. +impl wasmtime_wasi_http::p2::WasiHttpView for State { + fn http(&mut self) -> wasmtime_wasi_http::p2::WasiHttpCtxView<'_> { + wasmtime_wasi_http::p2::WasiHttpCtxView { + ctx: &mut self.http, + table: &mut self.table, + hooks: &mut self.egress, + } + } +} + +/// Convenience for callers that need a `Store` without naming `State`'s +/// internals. +pub fn new_store(engine: &wasmtime::Engine, state: State) -> Result> { + Ok(Store::new(engine, state)) +} + +#[cfg(test)] +mod tests { + use super::*; + use s3fs_core::backend::memory::MemoryBackend; + use s3fs_core::{Config, MasterSecret, OpenFlags}; + + /// A trapped guest's files are released by the runtime, not kept open + /// forever by the filesystem's handle table — and released *unflushed*. + /// The release runs after the tenant's next request may already have + /// committed, so a flush would land on top of that commit and undo it. + #[tokio::test] + async fn dropping_a_state_releases_its_files_without_flushing_them() { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([3; 32]), + [4; 16], + Arc::new(Config::default()), + ) + .await + .unwrap(); + let clock = Arc::new( + crate::clock::WallClockAdapter::new(Box::new(crate::clock::HostClock)).unwrap(), + ); + let mut state = State::new( + wasmtime_wasi::WasiCtxBuilder::new().build(), + fs.clone(), + clock, + ); + + let handle = fs + .open("/left-open", OpenFlags::create_new()) + .await + .unwrap(); + fs.pwrite(&handle, 0, b"OLD").await.unwrap(); + let id = handle.id; + state + .table + .push(Descriptor::File { + parent: fs.root(), + handle, + }) + .unwrap(); + drop(state); + + // The tenant's next request, which the lock lets in as soon as the + // dropped guest's call is over. + let next = fs + .open("/left-open", OpenFlags::read_write()) + .await + .unwrap(); + fs.pwrite(&next, 0, b"NEW").await.unwrap(); + fs.close(&next).await.unwrap(); + + for _ in 0..100 { + if fs.get_handle(id).is_none() { + break; + } + tokio::task::yield_now().await; + } + assert!(fs.get_handle(id).is_none(), "the handle is still open"); + let h = fs.open("/left-open", OpenFlags::read_only()).await.unwrap(); + assert_eq!( + fs.pread(&h, 0, 16).await.unwrap().as_ref(), + b"NEW", + "the dead guest's buffer landed on the next request's commit" + ); + } +} diff --git a/runtime/src/stream.rs b/runtime/src/stream.rs new file mode 100644 index 0000000..5d12cf5 --- /dev/null +++ b/runtime/src/stream.rs @@ -0,0 +1,1120 @@ +//! Outbound connections the runtime holds on a guest's behalf. +//! +//! # The problem this exists for +//! +//! A guest has no execution context of its own. An instance is created to serve +//! one invocation and dropped at the end — [`ServeHandle::verify_instantiates`] +//! says why in as many words — so it cannot own a socket that outlives the call, +//! and nothing inside it can reconnect, because between invocations there is no +//! "inside it". +//! +//! That left two ways to reach a guest, and neither serves a counterparty that +//! wants to *initiate*: +//! +//! - an inbound request, which the gate binds to a WebAuthn assertion, so the +//! caller must hold the tenant's passkey; +//! - the task queue, which is a clock. A timer can start work; it cannot be +//! spoken to. +//! +//! A service holding half of a threshold key has neither. It has no passkey for +//! the tenant whose money it is party to, and asking it to wait for the next +//! tick is not a conversation. +//! +//! So the runtime keeps the connection and the guest stays stateless: each +//! message the far side sends becomes one invocation, exactly as a task run is +//! one invocation. The guest replies from inside that call, and may send +//! unprompted from any invocation it is already in. +//! +//! # What it is on the wire +//! +//! Server-sent events for what arrives, `POST` for what is sent — two HTTP +//! shapes rather than one socket, because that is what the egress path already +//! carries and what a service can serve without a WebSocket stack. The logical +//! channel is bidirectional; the transport is ordinary. +//! +//! ```text +//! runtime ──GET /escrow/stream?id=…──▶ service held open, events arrive +//! runtime ──POST /escrow/send─────────▶ service one message, one request +//! │ +//! └── per event: invoke the guest's `on-message`, send back what it returns +//! ``` +//! +//! # What it does not promise +//! +//! Delivery is **at least once** in both directions: a reconnect may redeliver, +//! and a `POST` whose response was lost may have arrived. Handlers deduplicate +//! by their own identifiers, exactly as task handlers deduplicate by run id. +//! Ordering holds within one connection and not across a reconnect. +//! +//! # What bounds it +//! +//! Nothing, deliberately — that is the point. The *guest* invocation each +//! message causes is bounded like any other, but the connection itself is the +//! runtime's and outlives every one of them. A reconnect after a network +//! failure or a runtime restart needs no guest and no timer: the record is on +//! the filesystem, and the supervisor reads it at boot. + +use std::collections::BTreeMap; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; +use std::time::Duration; + +use anyhow::{bail, ensure, Context, Result}; +use s3fs_core::{Fs, Inode, OpenFlags}; +use serde::{Deserialize, Serialize}; +use tokio::sync::{Mutex, Notify}; + +use crate::serve::egress::Origin; +use crate::tenant::ensure_dir; + +const DIR: &str = "/runtime/streams"; +const MAX_RECORD: usize = 8 * 1024; +/// One message, in either direction. Large enough for a key package and a +/// transaction; small enough that a peer cannot make the runtime hold a lot. +pub const MAX_MESSAGE: usize = 256 * 1024; +/// How many connections one tenant may ask the runtime to hold. Each is a +/// socket and a supervisor; a tenant that could ask for unboundedly many could +/// exhaust the process on everyone else's behalf. +const PER_TENANT: usize = 8; + +/// Reconnection backoff. Starts quick, because the common failure is a service +/// restarting, and settles long, because the uncommon one is a service that is +/// gone and there is no point hammering it. +const BACKOFF: [Duration; 6] = [ + Duration::from_secs(1), + Duration::from_secs(2), + Duration::from_secs(5), + Duration::from_secs(15), + Duration::from_secs(60), + Duration::from_secs(300), +]; + +/// What is written down, and all of it but `generation`. A connection is a +/// standing instruction, not a session: there is no state worth keeping about +/// one that is currently up. +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] +pub struct StreamRecord { + pub version: u8, + #[serde(with = "hex_tenant")] + pub tenant: [u8; 16], + pub id: String, + /// Scheme, host and port. Admitted by the image's allowlist when opened and + /// again on every reconnect — a list narrowed between them takes effect. + pub origin: String, + /// Which `open` made this record; zero for one read back at boot. A close + /// and a reopen that land between two supervisor passes leave a record + /// equal in every field that is written down, and the pass must still see + /// a new instruction: the old connection belongs to the old one, and the + /// close took its status with it. + #[serde(skip)] + generation: u64, +} + +impl StreamRecord { + fn key(&self) -> String { + self.wire_id() + } + + /// What the far side is told this connection is called. + /// + /// The tenant, then the guest's own id — **not** the guest's id alone, and that is not + /// cosmetic. A stream id is tenant-*local*: a guest derives it from something about the + /// counterparty, so every tenant that guest serves opens a connection under the very same + /// name. A service seeing only that could not tell one customer's held connection from + /// another's, and — because a message sent to it arrives as a POST of its own, with no + /// connection identity in it — could not tell which connection a message belonged to either. + /// It would answer down whichever it happened to have, which is to say the wrong customer's. + /// + /// That is not hypothetical: it is what happened, and the symptom was one wallet's service + /// being told about an escrow a different wallet holds. + /// + /// The tenant is opaque and the far side already knows far more about who it is talking to — + /// it holds half of their escrow key — so this reveals nothing it did not have. + pub fn wire_id(&self) -> String { + format!("{}-{}", hex::encode(self.tenant), self.id) + } +} + +/// What a caller can see about a connection. Deliberately thin: whether it is +/// up, and enough history to tell "never worked" from "flapping". +#[derive(Debug, Clone, Serialize, Deserialize, Default)] +pub struct StreamStatus { + pub connected: bool, + pub connects: u64, + pub failures: u64, + pub last_error: Option, +} + +#[derive(Default)] +struct Live { + status: StreamStatus, +} + +/// The registry, and the supervisor that keeps its instructions true. +pub struct StreamRegistry { + fs: Arc, + dir: Arc, + records: Mutex>, + live: Mutex>, + /// Woken when a record appears or goes. `notify_one`, never `notify_waiters`: there is exactly + /// one consumer — the supervisor loop — and it is not registered as a waiter while it is + /// collecting records and spawning. `notify_waiters` would drop a wakeup that landed in that + /// window, and the connection would then wait out the poll below before it was ever dialled. + /// `notify_one` leaves a permit, so the next `notified()` returns at once. + changed: Notify, + egress: Mutex>, + messages: AtomicU64, + /// The next [`StreamRecord`] generation. Starts at 1, above every record + /// read back at boot. + generations: AtomicU64, +} + +/// What a guest's `enclave:streams/connection` calls are bound to. As with +/// tasks, the tenant comes from the invocation and is never a parameter. +#[derive(Clone)] +pub struct StreamContext { + pub registry: Arc, + pub tenant: [u8; 16], + /// Mutations require an interactive invocation, for the same reason task + /// mutations do: work arriving over a connection must not be able to grant + /// itself more connections. + pub interactive: bool, +} + +impl StreamRegistry { + pub async fn open_registry(fs: Arc) -> Result> { + let runtime = ensure_dir(&fs, &fs.root(), "runtime").await?; + let dir = ensure_dir(&fs, &runtime, "streams").await?; + let registry = Arc::new(Self { + fs, + dir, + records: Mutex::new(BTreeMap::new()), + live: Mutex::new(BTreeMap::new()), + changed: Notify::new(), + egress: Mutex::new(None), + messages: AtomicU64::new(0), + generations: AtomicU64::new(1), + }); + + // Everything standing, read back before anything is served. A record on + // disk is the whole of a connection's existence, so this is also the + // answer to "what reconnects after a restart": this loop. + let mut records = registry.records.lock().await; + for entry in registry.fs.read_dir(®istry.dir).await? { + if entry.name.ends_with(".tmp") { + registry.fs.unlink(®istry.dir, &entry.name).await?; + continue; + } + let h = registry + .fs + .open(&format!("{DIR}/{}", entry.name), OpenFlags::read_only()) + .await?; + let bytes = registry.fs.pread(&h, 0, MAX_RECORD + 1).await; + registry.fs.close(&h).await?; + let bytes = bytes?; + ensure!(bytes.len() <= MAX_RECORD, "oversized stream record"); + let record: StreamRecord = + serde_json::from_slice(&bytes).context("decoding a stream record")?; + ensure!( + record.version == 1 && valid_id(&record.id) && record.key() == entry.name, + "malformed stream record {}", + entry.name + ); + records.insert((record.tenant, record.id.clone()), record); + } + drop(records); + Ok(registry) + } + + /// The origins a guest may be connected to. Set once the image's policy is + /// known, and consulted on every connect rather than only at `open`. + pub async fn set_egress(&self, egress: crate::serve::EgressPolicy) { + *self.egress.lock().await = Some(egress); + } + + pub async fn open(&self, tenant: [u8; 16], id: String, origin: String) -> Result<()> { + ensure!(valid_id(&id), "a stream id is 1..=64 of [A-Za-z0-9_-]"); + let parsed = Origin::parse(&origin).context("that is not an origin")?; + self.admit(&parsed) + .await + .context("this image does not allow a connection to that origin")?; + + let mut records = self.records.lock().await; + if let Some(existing) = records.get(&(tenant, id.clone())) { + // Idempotent for the same instruction; an error for a different one, + // because silently moving a connection would move a conversation. + ensure!( + existing.origin == origin, + "a stream with that id is already open to {}", + existing.origin + ); + return Ok(()); + } + ensure!( + records.keys().filter(|(t, _)| *t == tenant).count() < PER_TENANT, + "a tenant may hold at most {PER_TENANT} connections" + ); + let record = StreamRecord { + version: 1, + tenant, + id, + origin, + generation: self.generations.fetch_add(1, Ordering::Relaxed), + }; + self.publish(&record).await?; + records.insert((record.tenant, record.id.clone()), record); + drop(records); + self.changed.notify_one(); + Ok(()) + } + + pub async fn close(&self, tenant: [u8; 16], id: &str) -> Result<()> { + let mut records = self.records.lock().await; + if let Some(record) = records.remove(&(tenant, id.to_string())) { + match self.fs.unlink(&self.dir, &record.key()).await { + Ok(()) | Err(s3fs_core::FsError::NotFound) => {} + Err(e) => return Err(e.into()), + } + } + drop(records); + self.live.lock().await.remove(&(tenant, id.to_string())); + self.changed.notify_one(); + Ok(()) + } + + pub async fn status(&self, tenant: [u8; 16], id: &str) -> Result { + let records = self.records.lock().await; + ensure!( + records.contains_key(&(tenant, id.to_string())), + "no such stream" + ); + drop(records); + Ok(self + .live + .lock() + .await + .get(&(tenant, id.to_string())) + .map(|l| l.status.clone()) + .unwrap_or_default()) + } + + /// One message to the far side. + /// + /// A plain request, not a write into the held connection: the held one + /// carries what arrives. Failing when the connection is down is deliberate — + /// only the caller knows whether a message is still worth sending when the + /// far side has not been there. + pub async fn send(&self, tenant: [u8; 16], id: &str, payload: Vec) -> Result<()> { + ensure!( + payload.len() <= MAX_MESSAGE, + "message exceeds {MAX_MESSAGE} bytes" + ); + let record = { + let records = self.records.lock().await; + records + .get(&(tenant, id.to_string())) + .cloned() + .context("no such stream")? + }; + let connected = self + .live + .lock() + .await + .get(&(tenant, id.to_string())) + .is_some_and(|l| l.status.connected); + ensure!( + connected, + "that stream is not connected; the message was not sent" + ); + + let origin = Origin::parse(&record.origin)?; + let egress = self.admit(&origin).await?; + let held = egress + .send_direct( + origin, + hyper::Request::builder() + .method(hyper::Method::POST) + .uri(format!( + "{}/escrow/send?id={}", + record.origin, + record.wire_id() + )) + .header("content-type", "application/octet-stream") + .body(body_from(payload))?, + Duration::from_secs(30), + ) + .await + .map_err(|e| anyhow::anyhow!("sending on {id}: {e:?}"))?; + ensure!( + held.response.status().is_success(), + "the far side refused a message: {}", + held.response.status() + ); + Ok(()) + } + + /// The allowlist, if it admits `origin` — checked on every connect and every + /// send, not only at `open`, so a list narrowed by a redeploy takes effect on + /// connections that were already standing. + async fn admit(&self, origin: &Origin) -> Result> { + match self.egress.lock().await.clone() { + Some(crate::serve::EgressPolicy::Allowlist(list)) + if list.origins().any(|o| o == origin) => + { + Ok(list) + } + _ => bail!("{origin:?} is not an origin this image allows"), + } + } + + async fn publish(&self, record: &StreamRecord) -> Result<()> { + let bytes = serde_json::to_vec(record)?; + ensure!(bytes.len() <= MAX_RECORD, "stream record exceeds limit"); + let temp = format!("{}.tmp", record.key()); + match self.fs.unlink(&self.dir, &temp).await { + Ok(()) | Err(s3fs_core::FsError::NotFound) => {} + Err(e) => return Err(e.into()), + } + let h = self + .fs + .open(&format!("{DIR}/{temp}"), OpenFlags::create_new()) + .await?; + let write = self.fs.pwrite(&h, 0, &bytes).await; + let close = self.fs.close(&h).await; + write?; + close?; // close commits the contents before publication + self.fs + .rename(&self.dir, &temp, &self.dir, &record.key()) + .await?; + Ok(()) + } + + /// How many messages have been delivered to a guest. For tests and for an + /// operator wanting to know whether a connection is doing anything. + pub fn delivered(&self) -> u64 { + self.messages.load(Ordering::Relaxed) + } + + /// Keep every standing instruction true, for as long as the process lives. + /// + /// One supervisor per record, started when the record appears and stopped + /// when it goes. Each holds a connection, hands what arrives to the guest, + /// and — the part no guest could do for itself — comes back after a failure. + /// + /// **This is what reconnects.** Not sealed state, which only says a + /// connection should exist, and not a timer, which was the thing worth + /// removing. A supervisor is a live task in the runtime: when its connection + /// drops it waits out the backoff and dials again, whether or not anybody is + /// using the wallet, and it starts at boot from the records on disk. + pub async fn run(self: Arc, guest: Arc) -> Result<()> { + let mut live: BTreeMap<([u8; 16], String), (StreamRecord, tokio::task::JoinHandle<()>)> = + BTreeMap::new(); + loop { + let wanted: Vec = self.records.lock().await.values().cloned().collect(); + + // Start what is new, and restart what changed: a close and reopen + // that land between two passes here leave the same key naming a + // different record — another origin, or the same one from a later + // `open` — and the old supervisor knows nothing of it. + for record in &wanted { + let key = (record.tenant, record.id.clone()); + match live.get(&key) { + Some((r, h)) if r == record && !h.is_finished() => continue, + Some((_, h)) => h.abort(), + None => {} + } + let registry = self.clone(); + let guest = guest.clone(); + let record = record.clone(); + let task = { + let record = record.clone(); + tokio::task::spawn(async move { registry.supervise(guest, record).await }) + }; + live.insert(key, (record, task)); + } + + // Stop what is no longer asked for. Aborting drops the connection, + // which is the whole of "close". + live.retain(|key, (_, handle)| { + let keep = wanted.iter().any(|r| (r.tenant, r.id.clone()) == *key); + if !keep { + handle.abort(); + } + keep + }); + + // Nothing to poll: a record appearing or going notifies, and a + // supervisor that exits leaves a finished handle for the next pass. + tokio::select! { + _ = self.changed.notified() => {} + _ = tokio::time::sleep(Duration::from_secs(30)) => {} + } + } + } + + /// One connection, held and re-held. + async fn supervise( + self: &Arc, + guest: Arc, + record: StreamRecord, + ) { + let key = (record.tenant, record.id.clone()); + let mut failures = 0usize; + loop { + match self.connect_once(&guest, &record).await { + Ok(()) => { + // The far side closed cleanly. Not an error, but not a + // reason to hammer it either — a service that ends the + // stream every time should not become a busy loop. + failures = 0; + self.note(&key, |s| s.connected = false).await; + tokio::time::sleep(BACKOFF[0]).await; + } + Err(e) => { + let message = format!("{e:#}"); + tracing::debug!(id = %record.id, error = %message, "stream connection ended"); + self.note(&key, |s| { + s.connected = false; + s.failures += 1; + s.last_error = Some(message); + }) + .await; + let wait = BACKOFF[failures.min(BACKOFF.len() - 1)]; + failures += 1; + tokio::time::sleep(wait).await; + } + } + } + } + + /// Hold one connection until it ends, handing each event to the guest. + async fn connect_once( + self: &Arc, + guest: &Arc, + record: &StreamRecord, + ) -> Result<()> { + use http_body_util::BodyExt; + + let origin = Origin::parse(&record.origin)?; + let egress = self.admit(&origin).await?; + let request = hyper::Request::builder() + .method(hyper::Method::GET) + .uri(format!( + "{}/escrow/stream?id={}", + record.origin, + record.wire_id() + )) + .header("accept", "text/event-stream") + .body(empty_body())?; + + // A long first-byte timeout: a service with nothing to say yet is the + // normal case, not a broken one. + // + // `held` is kept for the whole of the read loop below, and that is not + // tidiness: it owns the task driving the connection, and dropping it + // ends the body. See [`crate::serve::egress::HeldResponse`]. + let mut held = egress + .send_direct(origin, request, Duration::from_secs(60)) + .await + .map_err(|e| anyhow::anyhow!("{e:?}"))?; + ensure!( + held.response.status().is_success(), + "the service refused the stream: {}", + held.response.status() + ); + + let key = (record.tenant, record.id.clone()); + self.note(&key, |s| { + s.connected = true; + s.connects += 1; + s.last_error = None; + }) + .await; + + let mut framer = SseFramer::default(); + while let Some(frame) = held.response.frame().await { + let frame = frame.map_err(|e| anyhow::anyhow!("reading the stream: {e:?}"))?; + let Some(chunk) = frame.data_ref() else { + continue; + }; + for event in framer.push(chunk)? { + self.messages.fetch_add(1, Ordering::Relaxed); + // The guest answers, and what it answers goes back. An error is + // NOT sent: a guest that failed has said nothing it wants the + // far side to act on, and the message will arrive again. + match guest + .run_message(self, record.tenant, &record.id, &event.id, event.data) + .await + { + Ok(reply) if !reply.is_empty() => { + if let Err(e) = self.send(record.tenant, &record.id, reply).await { + tracing::debug!(id = %record.id, error = %format!("{e:#}"), "reply not sent"); + } + } + Ok(_) => {} + Err(e) => { + tracing::debug!(id = %record.id, error = %format!("{e:#}"), "guest refused a message"); + } + } + } + } + Ok(()) + } + + async fn note(&self, key: &([u8; 16], String), f: impl FnOnce(&mut StreamStatus)) { + let mut live = self.live.lock().await; + f(&mut live.entry(key.clone()).or_default().status); + } +} + +/// Server-sent events, reassembled from however the bytes arrive. +/// +/// The same problem the cosigner's own SSE reader solves against arkd: a frame +/// boundary is not an event boundary, and an event is not complete until a blank +/// line. Kept here rather than shared because the runtime must not depend on a +/// guest's crates. +#[derive(Default)] +struct SseFramer { + buffer: String, +} + +struct SseEvent { + id: String, + data: Vec, +} + +impl SseFramer { + fn push(&mut self, chunk: &[u8]) -> Result> { + self.buffer + .push_str(std::str::from_utf8(chunk).context("a stream frame was not UTF-8")?); + // The spec admits CRLF, CR and LF as line endings. Fold them to LF + // here, holding back a trailing CR that may be half of a CRLF the + // next read completes. + if self.buffer.contains('\r') { + let hold = self.buffer.ends_with('\r'); + let mut norm = self.buffer.replace("\r\n", "\n").replace('\r', "\n"); + if hold { + norm.pop(); + norm.push('\r'); + } + self.buffer = norm; + } + ensure!( + self.buffer.len() <= MAX_MESSAGE * 2, + "an unterminated event exceeded the message ceiling" + ); + let mut out = Vec::new(); + while let Some(end) = self.buffer.find("\n\n") { + let block: String = self.buffer.drain(..end + 2).collect(); + let mut id = String::new(); + let mut data = String::new(); + for line in block.lines() { + if let Some(v) = line.strip_prefix("id:") { + id = v.trim().to_string(); + } else if let Some(v) = line.strip_prefix("data:") { + data.push_str(v.trim()); + } + } + if data.is_empty() { + continue; // a heartbeat, or a comment + } + let bytes = base64_decode(&data).context("event data was not base64")?; + ensure!( + bytes.len() <= MAX_MESSAGE, + "event exceeds the message ceiling" + ); + // An event with no id of its own gets one from its content, so a + // guest deduplicating by id still can. + if id.is_empty() { + id = hex::encode(&nitro_attestation::sha256(&bytes)[..16]); + } + out.push(SseEvent { id, data: bytes }); + } + Ok(out) + } +} + +fn base64_decode(s: &str) -> Result> { + use base64::Engine; + Ok(base64::engine::general_purpose::STANDARD.decode(s)?) +} + +fn empty_body() -> wasmtime_wasi_http::p2::body::HyperOutgoingBody { + use http_body_util::{BodyExt, Empty}; + Empty::::new() + .map_err(|e: std::convert::Infallible| match e {}) + .boxed_unsync() +} + +fn body_from(bytes: Vec) -> wasmtime_wasi_http::p2::body::HyperOutgoingBody { + use http_body_util::{BodyExt, Full}; + Full::new(bytes::Bytes::from(bytes)) + .map_err(|e: std::convert::Infallible| match e {}) + .boxed_unsync() +} + +/// The same alphabet a task id uses, and for the same reason: an id ends up in +/// a filename and in a URL. +pub fn valid_id(s: &str) -> bool { + (1..=64).contains(&s.chars().count()) + && s.chars() + .all(|c| c.is_ascii_alphanumeric() || c == '_' || c == '-') +} + +mod hex_tenant { + use serde::{Deserialize, Deserializer, Serializer}; + pub fn serialize(v: &[u8; 16], s: S) -> Result { + s.serialize_str(&hex::encode(v)) + } + pub fn deserialize<'de, D: Deserializer<'de>>(d: D) -> Result<[u8; 16], D::Error> { + let s = String::deserialize(d)?; + let bytes = hex::decode(&s).map_err(serde::de::Error::custom)?; + bytes + .try_into() + .map_err(|_| serde::de::Error::custom("a tenant id is 16 bytes")) + } +} + +/// The ABI is defined in wit/stream/stream.wit. No tenant id is accepted. +pub fn add_to_linker( + linker: &mut wasmtime::component::Linker, +) -> wasmtime::Result<()> { + use crate::state::State; + + fn context(state: &State, mutation: bool) -> Result { + let ctx = state + .streams + .clone() + .context("streams require an authenticated tenant")?; + ensure!( + !mutation || ctx.interactive, + "work arriving over a connection cannot open more connections" + ); + Ok(ctx) + } + + let mut c = linker.instance("enclave:streams/connection@0.1.0")?; + c.func_wrap_async("stream-open", |store, (id, origin): (String, String)| { + Box::new(async move { + let result = async move { + let ctx = context(store.data(), true)?; + ctx.registry.open(ctx.tenant, id, origin).await + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + })?; + c.func_wrap_async("stream-close", |store, (id,): (String,)| { + Box::new(async move { + let result = async move { + let ctx = context(store.data(), true)?; + ctx.registry.close(ctx.tenant, &id).await + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + })?; + // Not a mutation: replying to what arrived is the whole point, and a guest + // invoked BY a message must be able to answer it. + c.func_wrap_async("stream-send", |store, (id, payload): (String, Vec)| { + Box::new(async move { + let result = async move { + let ctx = context(store.data(), false)?; + ctx.registry.send(ctx.tenant, &id, payload).await + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + })?; + c.func_wrap_async("stream-status", |store, (id,): (String,)| { + Box::new(async move { + let result: Result = async move { + let ctx = context(store.data(), false)?; + Ok(serde_json::to_string( + &ctx.registry.status(ctx.tenant, &id).await?, + )?) + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + })?; + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use s3fs_core::backend::memory::MemoryBackend; + use s3fs_core::{Config, MasterSecret}; + + async fn registry() -> (Arc, Arc) { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend.clone(), + &MasterSecret::from_bytes([7; 32]), + [8; 16], + Arc::new(Config::default()), + ) + .await + .unwrap(); + crate::tenant::tenant_root_by_id(&fs, [1; 16]) + .await + .unwrap(); + let r = StreamRegistry::open_registry(fs.clone()).await.unwrap(); + r.set_egress(crate::serve::EgressPolicy::Allowlist(Arc::new( + crate::serve::egress::EgressAllowlist::parse(&["https://svc.example"]).unwrap(), + ))) + .await; + (r, fs) + } + + #[tokio::test] + async fn an_origin_the_image_does_not_allow_is_refused() { + let (r, _fs) = registry().await; + let err = r + .open([1; 16], "esc".into(), "https://elsewhere.example".into()) + .await + .unwrap_err(); + assert!(format!("{err:#}").contains("does not allow"), "{err:#}"); + } + + /// Two tenants opening a connection under the same local id must be distinguishable by the + /// far side, which sees only what is on the wire. + #[tokio::test] + async fn one_id_from_two_tenants_is_two_names_on_the_wire() { + let a = StreamRecord { + version: 1, + tenant: [1; 16], + id: "svc-abc".into(), + origin: "https://svc.example".into(), + generation: 0, + }; + let b = StreamRecord { + tenant: [2; 16], + ..a.clone() + }; + assert_ne!( + a.wire_id(), + b.wire_id(), + "a service cannot tell two customers apart if their connections share a name" + ); + assert!(a.wire_id().ends_with("-svc-abc"), "{}", a.wire_id()); + // And it is the same name the record is filed under, so there is one identity and not two. + assert_eq!(a.wire_id(), a.key()); + } + + /// A wakeup that lands while the supervisor loop is between iterations must not be lost. + /// + /// This is not hypothetical: `notify_waiters` dropped exactly this one, and the effect was a + /// guest opening a connection and finding it still down thirty seconds later — long past any + /// patience a call has. `notify_one` leaves a permit, so the wakeup keeps. + #[tokio::test] + async fn a_wakeup_that_lands_between_iterations_is_not_lost() { + let (r, _fs) = registry().await; + // Nobody is waiting on `changed` yet — the supervisor loop is not running at all, which is + // the strongest version of "between iterations". + r.open([1; 16], "svc".into(), "https://svc.example".into()) + .await + .unwrap(); + + // The wakeup is still there to be collected. + tokio::time::timeout(Duration::from_millis(50), r.changed.notified()) + .await + .expect("the wakeup from `open` must survive until somebody waits for it"); + } + + #[tokio::test] + async fn opening_the_same_instruction_twice_is_a_no_op_and_a_different_one_is_an_error() { + let (r, _fs) = registry().await; + r.open([1; 16], "esc".into(), "https://svc.example".into()) + .await + .unwrap(); + r.open([1; 16], "esc".into(), "https://svc.example".into()) + .await + .expect("the same instruction again changes nothing"); + + r.set_egress(crate::serve::EgressPolicy::Allowlist(Arc::new( + crate::serve::egress::EgressAllowlist::parse(&[ + "https://svc.example", + "https://other.example", + ]) + .unwrap(), + ))) + .await; + let err = r + .open([1; 16], "esc".into(), "https://other.example".into()) + .await + .unwrap_err(); + assert!(format!("{err:#}").contains("already open"), "{err:#}"); + } + + /// The answer to "what reconnects after a restart": the record, read back. + #[tokio::test] + async fn a_connection_survives_a_fresh_mount() { + let (r, fs) = registry().await; + r.open([1; 16], "esc".into(), "https://svc.example".into()) + .await + .unwrap(); + drop(r); + + let reopened = StreamRegistry::open_registry(fs).await.unwrap(); + let records = reopened.records.lock().await; + let record = records + .get(&([1; 16], "esc".to_string())) + .expect("the instruction came back with no guest involved"); + assert_eq!(record.origin, "https://svc.example"); + } + + #[tokio::test] + async fn closing_forgets_it_and_is_idempotent() { + let (r, fs) = registry().await; + r.open([1; 16], "esc".into(), "https://svc.example".into()) + .await + .unwrap(); + r.close([1; 16], "esc").await.unwrap(); + r.close([1; 16], "esc") + .await + .expect("closing twice is fine"); + + let reopened = StreamRegistry::open_registry(fs).await.unwrap(); + assert!(reopened.records.lock().await.is_empty()); + } + + #[tokio::test] + async fn a_tenant_cannot_hold_unboundedly_many() { + let (r, _fs) = registry().await; + for i in 0..PER_TENANT { + r.open([1; 16], format!("esc{i}"), "https://svc.example".into()) + .await + .unwrap(); + } + let err = r + .open([1; 16], "one-too-many".into(), "https://svc.example".into()) + .await + .unwrap_err(); + assert!(format!("{err:#}").contains("at most"), "{err:#}"); + + // Another tenant is unaffected: the cap is per tenant, not global. + crate::tenant::tenant_root_by_id(&r.fs, [2; 16]) + .await + .unwrap(); + r.open([2; 16], "esc".into(), "https://svc.example".into()) + .await + .expect("one tenant's quota is not another's"); + } + + #[tokio::test] + async fn sending_on_a_stream_that_is_down_says_so_rather_than_dropping_it() { + let (r, _fs) = registry().await; + r.open([1; 16], "esc".into(), "https://svc.example".into()) + .await + .unwrap(); + let err = r.send([1; 16], "esc", vec![1, 2, 3]).await.unwrap_err(); + assert!(format!("{err:#}").contains("not connected"), "{err:#}"); + } + + // --- Framing ------------------------------------------------------------- + // + // A frame boundary is not an event boundary. These are the cases that would + // otherwise show up as a guest being handed half a message. + + fn b64(bytes: &[u8]) -> String { + use base64::Engine; + base64::engine::general_purpose::STANDARD.encode(bytes) + } + + #[test] + fn an_event_split_across_reads_is_reassembled() { + let mut f = SseFramer::default(); + let whole = format!("id: m1\ndata: {}\n\n", b64(b"hello")); + let (a, b) = whole.split_at(9); + assert!( + f.push(a.as_bytes()).unwrap().is_empty(), + "half an event is no event" + ); + let out = f.push(b.as_bytes()).unwrap(); + assert_eq!(out.len(), 1); + assert_eq!(out[0].id, "m1"); + assert_eq!(out[0].data, b"hello"); + } + + /// The spec admits CRLF and CR as line endings, and a real server sends + /// them — including a CRLF cut between two reads. + #[test] + fn crlf_and_cr_terminated_events_are_delivered() { + let mut f = SseFramer::default(); + let crlf = format!("id: m1\r\ndata: {}\r\n\r\n", b64(b"one")); + let (a, b) = crlf.split_at(crlf.len() - 3); // split inside the final CRLF + assert!(f.push(a.as_bytes()).unwrap().is_empty()); + let out = f.push(b.as_bytes()).unwrap(); + assert_eq!(out.len(), 1); + assert_eq!(out[0].id, "m1"); + assert_eq!(out[0].data, b"one"); + + // A chunk-final CR is held back — it may be half a CRLF — so the + // CR-terminated event is followed by the start of the next line. + let cr = format!("data: {}\r\r: hb\r", b64(b"two")); + let out = f.push(cr.as_bytes()).unwrap(); + assert_eq!(out.len(), 1); + assert_eq!(out[0].data, b"two"); + } + + #[test] + fn several_events_in_one_read_all_come_out_in_order() { + let mut f = SseFramer::default(); + let chunk = format!( + "id: a\ndata: {}\n\nid: b\ndata: {}\n\n", + b64(b"one"), + b64(b"two") + ); + let out = f.push(chunk.as_bytes()).unwrap(); + assert_eq!(out.len(), 2); + assert_eq!(out[0].data, b"one"); + assert_eq!(out[1].data, b"two"); + } + + /// Heartbeats keep a quiet connection alive and are not messages. + #[test] + fn a_heartbeat_is_not_delivered_to_the_guest() { + let mut f = SseFramer::default(); + assert!(f.push(b": keep-alive\n\n").unwrap().is_empty()); + assert!(f.push(b"\n\n").unwrap().is_empty()); + } + + /// A guest deduplicates by id, so an event without one still needs a stable + /// handle — the same bytes must give the same id. + #[test] + fn an_event_with_no_id_gets_a_stable_one_from_its_content() { + let mut f = SseFramer::default(); + let first = f + .push(format!("data: {}\n\n", b64(b"x")).as_bytes()) + .unwrap(); + let second = f + .push(format!("data: {}\n\n", b64(b"x")).as_bytes()) + .unwrap(); + let third = f + .push(format!("data: {}\n\n", b64(b"y")).as_bytes()) + .unwrap(); + assert_eq!(first[0].id, second[0].id, "the same message is the same id"); + assert_ne!(first[0].id, third[0].id); + } + + #[test] + fn an_unterminated_event_cannot_grow_without_limit() { + let mut f = SseFramer::default(); + // The buffer must admit one whole event — base64 is about 4/3 the bytes, + // plus its field names — so the ceiling is above MAX_MESSAGE and a test + // for it has to actually exceed it. + let big = "d".repeat(MAX_MESSAGE); + assert!( + f.push(big.as_bytes()).is_ok(), + "one max-size event must fit" + ); + let err = loop { + match f.push(big.as_bytes()) { + Ok(_) => continue, + Err(e) => break e, + } + }; + assert!( + format!("{err:#}").contains("unterminated"), + "a peer that never terminates an event must not be able to fill memory: {err:#}" + ); + } + + #[test] + fn data_that_is_not_base64_is_an_error_rather_than_a_guess() { + let mut f = SseFramer::default(); + assert!(f.push(b"id: m\ndata: not base64!!\n\n").is_err()); + } + + #[tokio::test] + async fn an_id_that_would_not_be_a_filename_is_refused() { + let (r, _fs) = registry().await; + for bad in ["", "../escape", "has space", &"x".repeat(65)] { + assert!( + r.open([1; 16], bad.into(), "https://svc.example".into()) + .await + .is_err(), + "{bad:?} should not be a stream id" + ); + } + } + + async fn guest(fs: Arc) -> Arc { + let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-http/target/wasm32-wasip2/release/guest-http.wasm"); + let bytes = std::fs::read(path).expect("build examples/guest-http for wasm32-wasip2"); + let (logs, _collector) = crate::guest_io::start(Arc::new(crate::TracingLogSink)); + let env = crate::GuestEnvironment::new( + fs, + Box::new(crate::clock::HostClock), + Arc::new(nitro_nsm::fake::FakeNsm::new()), + &[], + &[], + logs, + ) + .unwrap(); + let engine = crate::ServeHandle::engine_with_watchdog().unwrap(); + Arc::new(crate::ServeHandle::new(&engine, &bytes, env).unwrap()) + } + + /// Accept one connection and hold it open as an event stream. + async fn serve_one(listener: &tokio::net::TcpListener) -> tokio::net::TcpStream { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + let (mut socket, _) = listener.accept().await.unwrap(); + let mut request = [0; 8192]; + let _ = socket.read(&mut request).await.unwrap(); + socket + .write_all(b"HTTP/1.1 200 OK\r\ncontent-type: text/event-stream\r\n\r\n: hello\n\n") + .await + .unwrap(); + socket + } + + /// A reopen gets a connection of its own, whether to a new origin or to + /// the same one. + /// + /// The same-origin half is the one that broke. A close and a reopen + /// between two supervisor passes left a record equal to the old one, so + /// the old connection stayed up — with its status gone, which made `send` + /// refuse on it, and with it every reply to a message that arrived. + #[tokio::test] + #[ignore = "requires built guest-http component"] + async fn a_reopened_stream_gets_a_new_connection() { + let a = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let b = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let origin_a = format!("http://{}", a.local_addr().unwrap()); + let origin_b = format!("http://{}", b.local_addr().unwrap()); + let (r, fs) = registry().await; + r.set_egress(crate::serve::EgressPolicy::Allowlist(Arc::new( + crate::serve::egress::EgressAllowlist::parse(&[origin_a.as_str(), origin_b.as_str()]) + .unwrap(), + ))) + .await; + let supervisor = tokio::spawn(r.clone().run(guest(fs).await)); + let wait = Duration::from_secs(5); + + r.open([1; 16], "esc".into(), origin_a.clone()) + .await + .unwrap(); + let _first = tokio::time::timeout(wait, serve_one(&a)) + .await + .expect("never connected"); + + r.close([1; 16], "esc").await.unwrap(); + r.open([1; 16], "esc".into(), origin_a).await.unwrap(); + let _second = tokio::time::timeout(wait, serve_one(&a)) + .await + .expect("reopened to the same origin, and the old connection was kept"); + + r.close([1; 16], "esc").await.unwrap(); + r.open([1; 16], "esc".into(), origin_b).await.unwrap(); + let _third = tokio::time::timeout(wait, serve_one(&b)) + .await + .expect("reopened to a new origin, and it was never dialled"); + + supervisor.abort(); + } +} diff --git a/runtime/src/tasks.rs b/runtime/src/tasks.rs new file mode 100644 index 0000000..bf88d45 --- /dev/null +++ b/runtime/src/tasks.rs @@ -0,0 +1,1188 @@ +//! Durable, tenant-bound background work. One queue owner per mounted filesystem. +//! +//! A synced temporary file is atomically renamed to publish each record. The +//! record itself is the scheduling index; notifications are only an optimization. +//! Recovery tolerates unpublished temporary files, never malformed live records. +//! Execution is at least once: handlers must deduplicate their effects by run ID. +use std::collections::{BTreeMap, BTreeSet}; +use std::sync::Arc; +use std::time::Duration; + +use anyhow::{ensure, Context, Result}; +use s3fs_core::{Fs, Inode, OpenFlags}; +use serde::{Deserialize, Serialize}; +use tokio::sync::{Mutex, Notify}; +use wasmtime_wasi::HostWallClock; + +use crate::clock::WallClockAdapter; +use crate::state::State; +use crate::tenant::ensure_dir; + +const MAX_PAYLOAD: usize = 64 * 1024; +const MAX_RECORD: usize = 1024 * 1024; +const MAX_ATTEMPTS: u32 = 5; +const DIR: &str = "/runtime/tasks"; + +#[derive(Debug, Clone)] +pub struct TaskLimits { + pub concurrency: usize, + pub max_records: usize, + pub per_tenant: usize, + pub timeout: Duration, +} +impl Default for TaskLimits { + fn default() -> Self { + Self { + concurrency: 1, + max_records: 1024, + per_tenant: 64, + timeout: Duration::from_secs(30), + } + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum Status { + Pending, + Running, + Completed, + Failed, + Cancelled, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct Task { + pub version: u32, + pub tenant: [u8; 16], + pub id: String, + pub payload: Vec, + pub run_at: u64, + pub requested_run_at: u64, + pub generation: [u8; 16], + pub interval_ms: Option, + pub status: Status, + pub attempts: u32, + pub occurrence: u64, + pub result: Vec, + pub error: Option, +} +impl Task { + pub fn run_id(&self) -> String { + format!( + "{}:{}:{}", + self.id, + hex::encode(self.generation), + self.occurrence + ) + } + fn key(&self) -> String { + key(self.tenant, &self.id) + } + fn terminal(&self) -> bool { + matches!( + self.status, + Status::Completed | Status::Failed | Status::Cancelled + ) + } +} +fn key(tenant: [u8; 16], id: &str) -> String { + format!("{}-{id}", hex::encode(tenant)) +} +fn valid_id(id: &str) -> bool { + !id.is_empty() + && id.len() <= 64 + && id + .bytes() + .all(|b| b.is_ascii_alphanumeric() || b == b'-' || b == b'_') +} + +#[derive(Default)] +struct Records { + tasks: BTreeMap, + due: BTreeSet<(u64, String)>, + active: BTreeSet, + last_tenant: Option<[u8; 16]>, +} +pub struct TaskQueue { + fs: Arc, + dir: Arc, + clock: Arc, + pub limits: TaskLimits, + records: Mutex, + changed: Notify, +} +impl TaskQueue { + pub async fn open( + fs: Arc, + clock: Arc, + limits: TaskLimits, + ) -> Result> { + ensure!( + limits.concurrency > 0 + && limits.max_records > 0 + && limits.per_tenant > 0 + && !limits.timeout.is_zero(), + "task limits must be positive" + ); + let runtime = ensure_dir(&fs, &fs.root(), "runtime").await?; + let dir = ensure_dir(&fs, &runtime, "tasks").await?; + let queue = Arc::new(Self { + fs, + dir, + clock, + limits, + records: Mutex::new(Records::default()), + changed: Notify::new(), + }); + let mut records = queue.records.lock().await; + for entry in queue.fs.read_dir(&queue.dir).await? { + if entry.name.ends_with(".tmp") { + queue.fs.unlink(&queue.dir, &entry.name).await?; + continue; + } + ensure!( + records.tasks.len() < queue.limits.max_records, + "task store exceeds configured quota" + ); + let h = queue + .fs + .open(&format!("{DIR}/{}", entry.name), OpenFlags::read_only()) + .await?; + let bytes = queue.fs.pread(&h, 0, MAX_RECORD + 1).await; + queue.fs.close(&h).await?; + let bytes = bytes?; + ensure!(bytes.len() <= MAX_RECORD, "oversized task record"); + let mut task: Task = serde_json::from_slice(&bytes).context("decoding task record")?; + ensure!( + task.version == 1 + && valid_id(&task.id) + && task.key() == entry.name + && task.payload.len() <= MAX_PAYLOAD + && task.result.len() <= MAX_PAYLOAD + && task.interval_ms.is_none_or(|i| i >= 1000), + "invalid task record" + ); + // Retry interrupted attempts with their original occurrence ID. + if task.status == Status::Running { + task.status = if task.attempts >= MAX_ATTEMPTS { + Status::Failed + } else { + Status::Pending + }; + task.error = Some("enclave stopped during execution".into()); + queue.write(&task).await?; + tracing::warn!(tenant = %hex::encode(task.tenant), task = %task.id, + status = ?task.status, attempts = task.attempts, + "background task attempt interrupted by a restart"); + } + if task.status == Status::Pending { + // Recovery jitter prevents a fleet of overdue tasks firing together. + let due = task + .run_at + .max(queue.now().saturating_add(offset(&task.key(), 1000))); + records.due.insert((due, task.key())); + } + records.tasks.insert(task.key(), task); + } + drop(records); + Ok(queue) + } + pub fn now(&self) -> u64 { + self.clock.now().as_millis().min(u64::MAX as u128) as u64 + } + async fn write(&self, task: &Task) -> Result<()> { + let bytes = serde_json::to_vec(task)?; + ensure!(bytes.len() <= MAX_RECORD, "task record exceeds limit"); + let temp = format!("{}.tmp", task.key()); + // Remove an unpublished write left by a cancelled host call. + match self.fs.unlink(&self.dir, &temp).await { + Ok(()) | Err(s3fs_core::FsError::NotFound) => {} + Err(e) => return Err(e.into()), + } + let h = self + .fs + .open(&format!("{DIR}/{temp}"), OpenFlags::create_new()) + .await?; + let write = self.fs.pwrite(&h, 0, &bytes).await; + let close = self.fs.close(&h).await; + write?; + close?; // close commits the complete contents before publication + self.fs + .rename(&self.dir, &temp, &self.dir, &task.key()) + .await?; + Ok(()) + } + pub async fn enqueue( + &self, + tenant: [u8; 16], + id: String, + payload: Vec, + run_at: u64, + interval_ms: Option, + ) -> Result<()> { + ensure!(valid_id(&id), "invalid task id"); + ensure!(payload.len() <= MAX_PAYLOAD, "task payload exceeds 64 KiB"); + ensure!( + interval_ms.is_none_or(|i| i >= 1000), + "interval must be at least 1000 ms" + ); + crate::tenant::existing_tenant_root(&self.fs, tenant).await?; + let mut r = self.records.lock().await; + let k = key(tenant, &id); + if let Some(old) = r.tasks.get(&k) { + ensure!( + old.payload == payload + && old.interval_ms == interval_ms + && old.requested_run_at == run_at, + "task id already used with different input" + ); + return Ok(()); + } + ensure!( + r.tasks.len() < self.limits.max_records + && r.tasks.values().filter(|t| t.tenant == tenant).count() < self.limits.per_tenant, + "task quota reached; forget terminal tasks" + ); + let mut generation = [0; 16]; + getrandom::fill(&mut generation) + .map_err(|e| anyhow::anyhow!("task identity entropy: {e}"))?; + let task = Task { + version: 1, + tenant, + id, + payload, + run_at, + requested_run_at: run_at, + generation, + interval_ms, + status: Status::Pending, + attempts: 0, + occurrence: 0, + result: vec![], + error: None, + }; + self.write(&task).await?; + r.due.insert((run_at, k.clone())); + r.tasks.insert(k, task); + self.changed.notify_one(); + Ok(()) + } + pub async fn status(&self, tenant: [u8; 16], id: &str) -> Result { + self.records + .lock() + .await + .tasks + .get(&key(tenant, id)) + .cloned() + .context("no such task") + } + pub async fn cancel(&self, tenant: [u8; 16], id: &str) -> Result<()> { + let mut r = self.records.lock().await; + let k = key(tenant, id); + let mut task = r.tasks.get(&k).cloned().context("no such task")?; + task.status = Status::Cancelled; + self.write(&task).await?; + r.tasks.insert(k.clone(), task); + r.due.retain(|(_, candidate)| candidate != &k); + self.changed.notify_one(); + Ok(()) + } + pub async fn forget(&self, tenant: [u8; 16], id: &str) -> Result<()> { + let mut r = self.records.lock().await; + let k = key(tenant, id); + ensure!( + !r.active.contains(&k) && r.tasks.get(&k).context("no such task")?.terminal(), + "task is still active" + ); + self.fs.unlink(&self.dir, &k).await?; + r.tasks.remove(&k); + Ok(()) + } + async fn due(&self) -> Option { + let mut r = self.records.lock().await; + let now = self.now(); + // Deadlines decide eligibility; ready tenants then take turns. A + // tenant with a backlog of old tasks must not monopolize the worker + // until that entire backlog has drained. Inspect only due records, + // bounded by max_records, and preserve deadline order within a tenant. + let ((at, k), tenant) = r + .due + .iter() + .take_while(|(at, _)| *at <= now) + .filter_map(|entry| r.tasks.get(&entry.1).map(|task| (entry, task.tenant))) + .min_by_key(|((at, _), tenant)| { + ( + r.last_tenant.is_some_and(|last| *tenant <= last), + *tenant, + *at, + ) + }) + .map(|(entry, tenant)| (entry.clone(), tenant))?; + r.due.remove(&(at, k.clone())); + r.last_tenant = Some(tenant); + r.active.insert(k.clone()); + r.tasks.get(&k).cloned() + } + pub(crate) async fn begin(&self, task: &Task) -> Result { + let mut r = self.records.lock().await; + let mut current = r + .tasks + .get(&task.key()) + .cloned() + .context("task disappeared")?; + if current.status != Status::Pending { + return Ok(false); + } + current.status = Status::Running; + current.attempts += 1; + self.write(¤t).await?; + r.tasks.insert(current.key(), current); + Ok(true) + } + async fn finish(&self, task: &Task, outcome: Result>>) -> Result<()> { + let mut r = self.records.lock().await; + let k = task.key(); + let mut current = r.tasks.get(&k).cloned().context("task disappeared")?; + r.active.remove(&k); + if current.status == Status::Cancelled { + return Ok(()); + } + if matches!(outcome, Ok(None)) { + // A busy tenant never spends a retry or writes another record. + r.due.insert((self.now().saturating_add(100), k)); + return Ok(()); + } + ensure!( + current.status == Status::Running, + "task could not persist its execution intent" + ); + match outcome { + Ok(Some(bytes)) => { + current.result = bytes; + current.error = None; + if let Some(interval) = current.interval_ms { + current.status = Status::Pending; + current.attempts = 0; + current.occurrence = current + .occurrence + .checked_add(1) + .context("occurrence overflow")?; + // First full interval after now, with a stable per-job phase. + let now = self.now(); + let phase = offset(&k, interval); + let position = now % interval; + let delay = if phase > position { + phase - position + } else { + interval - (position - phase) + }; + current.run_at = now.saturating_add(delay); + } else { + current.status = Status::Completed; + } + } + Err(e) => { + // `{:#}` keeps the causes: a trap or a deadline is otherwise + // recorded as its outermost context and nothing of why. + current.error = Some(format!("{e:#}").chars().take(512).collect()); + current.status = if current.attempts >= MAX_ATTEMPTS { + Status::Failed + } else { + Status::Pending + }; + current.run_at = self + .now() + .saturating_add(1000u64 << current.attempts.min(8)); + } + Ok(None) => unreachable!(), + } + self.write(¤t).await?; + // A failure is logged with its reason. The record is sealed, so + // otherwise the only thing an operator sees is that five attempts + // failed. The text is what the guest could already print to its own + // log; `?` escapes it so it cannot forge lines of its own. + match ¤t.error { + Some(error) => tracing::warn!(tenant = %hex::encode(current.tenant), + task = %current.id, status = ?current.status, attempts = current.attempts, + error = ?error, "background task attempt failed"), + None => tracing::info!(tenant = %hex::encode(current.tenant), task = %current.id, + status = ?current.status, attempts = current.attempts, + "background task recorded"), + } + if current.status == Status::Pending { + r.due.insert((current.run_at, k.clone())); + } + r.tasks.insert(k, current); + Ok(()) + } + pub async fn run(self: Arc, guest: Arc) -> Result<()> { + let mut workers = tokio::task::JoinSet::new(); + loop { + while workers.len() < self.limits.concurrency { + let Some(task) = self.due().await else { + break; + }; + let queue = self.clone(); + let guest = guest.clone(); + workers.spawn(async move { + let result = guest.run_background(&queue, &task).await; + queue.finish(&task, result).await + }); + } + let next = self.records.lock().await.due.first().map(|(at, _)| *at); + let wait = Duration::from_millis( + next.map(|at| at.saturating_sub(self.now()).max(1)) + .unwrap_or(60_000) + .min(60_000), + ); + tokio::select! { + result = workers.join_next(), if !workers.is_empty() => { result.context("worker missing")???; }, + _ = self.changed.notified() => {}, + _ = tokio::time::sleep(wait), if workers.len() < self.limits.concurrency => {}, + } + } + } +} +fn offset(key: &str, interval: u64) -> u64 { + let hash = nitro_attestation::sha256(key.as_bytes()); + u64::from_le_bytes(hash[..8].try_into().unwrap()) % interval +} + +#[derive(Clone)] +pub struct TaskContext { + pub queue: Arc, + pub tenant: [u8; 16], + pub interactive: bool, +} +fn context(state: &State, mutation: bool) -> Result { + let ctx = state + .tasks + .clone() + .context("tasks require an authenticated tenant and enabled scheduler")?; + ensure!( + !mutation || ctx.interactive, + "background work cannot authorize more work" + ); + Ok(ctx) +} +/// The ABI is defined in wit/tasks/tasks.wit. No tenant ID is accepted. +pub fn add_to_linker(linker: &mut wasmtime::component::Linker) -> wasmtime::Result<()> { + let mut queue = linker.instance("enclave:tasks/queue@0.1.0")?; + queue.func_wrap_async( + "enqueue", + |store, (id, payload, at, interval): (String, Vec, u64, Option)| { + Box::new(async move { + let result = async move { + let ctx = context(store.data(), true)?; + ctx.queue + .enqueue(ctx.tenant, id, payload, at, interval) + .await + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + }, + )?; + queue.func_wrap_async("status", |store, (id,): (String,)| { + Box::new(async move { + let result: Result = async move { + let ctx = context(store.data(), false)?; + Ok(serde_json::to_string( + &ctx.queue.status(ctx.tenant, &id).await?, + )?) + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + })?; + queue.func_wrap_async("cancel", |store, (id,): (String,)| { + Box::new(async move { + let result = async move { + let ctx = context(store.data(), true)?; + ctx.queue.cancel(ctx.tenant, &id).await + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + })?; + queue.func_wrap_async("forget", |store, (id,): (String,)| { + Box::new(async move { + let result = async move { + let ctx = context(store.data(), true)?; + ctx.queue.forget(ctx.tenant, &id).await + } + .await; + Ok((result.map_err(|e| e.to_string()),)) + }) + })?; + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::clock::{HostClock, TrustedClock}; + use s3fs_core::backend::memory::MemoryBackend; + use s3fs_core::{Config, MasterSecret}; + use std::sync::atomic::{AtomicU64, Ordering}; + + #[derive(Debug, Clone)] + struct Clock(Arc); + impl TrustedClock for Clock { + fn now(&self) -> Result { + Ok(Duration::from_millis(self.0.load(Ordering::Relaxed))) + } + fn resolution(&self) -> Duration { + Duration::from_millis(1) + } + fn describe(&self) -> String { + "test".into() + } + } + async fn queue(limits: TaskLimits) -> (Arc, Clock) { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([9; 32]), + [1; 16], + Arc::new(Config::default()), + ) + .await + .unwrap(); + for tenant in [[1; 16], [2; 16]] { + crate::tenant::tenant_root_by_id(&fs, tenant).await.unwrap(); + } + let clock = Clock(Arc::new(AtomicU64::new(1000))); + let queue = TaskQueue::open( + fs, + Arc::new(WallClockAdapter::new(Box::new(clock.clone())).unwrap()), + limits, + ) + .await + .unwrap(); + (queue, clock) + } + async fn add(q: &TaskQueue, tenant: u8, id: &str, at: u64) { + q.enqueue([tenant; 16], id.into(), vec![tenant], at, None) + .await + .unwrap(); + } + #[tokio::test] + async fn identity_quota_and_idempotency_are_tenant_scoped() { + let (q, _) = queue(TaskLimits { + per_tenant: 1, + ..Default::default() + }) + .await; + add(&q, 1, "same", 0).await; + add(&q, 1, "same", 0).await; + add(&q, 2, "same", 0).await; + assert!(q + .enqueue([1; 16], "same".into(), vec![2], 0, None) + .await + .is_err()); + assert!(q + .enqueue([1; 16], "same".into(), vec![1], 5, None) + .await + .is_err()); + assert!(q + .enqueue([1; 16], "other".into(), vec![], 0, None) + .await + .is_err()); + assert_eq!(q.status([2; 16], "same").await.unwrap().payload, vec![2]); + assert!(q.cancel([3; 16], "same").await.is_err()); + assert!(q + .enqueue([3; 16], "new".into(), vec![], 0, None) + .await + .is_err()); + q.cancel([1; 16], "same").await.unwrap(); + q.forget([1; 16], "same").await.unwrap(); + add(&q, 1, "other", 0).await; + assert!(q + .enqueue([2; 16], "../escape".into(), vec![], 0, None) + .await + .is_err()); + } + #[tokio::test] + async fn ready_tenants_take_turns_instead_of_draining_one_backlog() { + let (q, _) = queue(Default::default()).await; + add(&q, 1, "old-a", 0).await; + add(&q, 1, "old-b", 0).await; + add(&q, 2, "newer", 500).await; + assert_eq!(q.due().await.unwrap().tenant, [1; 16]); + assert_eq!(q.due().await.unwrap().tenant, [2; 16]); + assert_eq!(q.due().await.unwrap().tenant, [1; 16]); + } + + #[tokio::test] + async fn interrupted_execution_recovers_with_the_same_run_id() { + let (q, clock) = queue(Default::default()).await; + add(&q, 1, "job", 0).await; + let task = q.due().await.unwrap(); + assert!(q.begin(&task).await.unwrap()); + let fs = q.fs.clone(); + let timer = q.clock.clone(); + drop(q); + let recovered = TaskQueue::open(fs, timer, Default::default()) + .await + .unwrap(); + clock.0.store(10_000, Ordering::Relaxed); + let again = recovered.due().await.unwrap(); + assert_eq!(again.run_id(), task.run_id()); + assert_eq!(again.attempts, 1); + recovered.begin(&again).await.unwrap(); + recovered + .finish(&again, Ok(Some(b"done".to_vec()))) + .await + .unwrap(); + let result = recovered.status([1; 16], "job").await.unwrap(); + assert_eq!(result.status, Status::Completed); + assert_eq!(result.result, b"done"); + let reopened = TaskQueue::open( + recovered.fs.clone(), + recovered.clock.clone(), + Default::default(), + ) + .await + .unwrap(); + assert!(reopened.due().await.is_none()); + assert_eq!( + reopened.status([1; 16], "job").await.unwrap().result, + b"done" + ); + } + #[tokio::test] + async fn cancellation_wins_over_an_inflight_completion() { + let (q, _) = queue(Default::default()).await; + add(&q, 1, "job", 0).await; + let task = q.due().await.unwrap(); + q.begin(&task).await.unwrap(); + q.cancel([1; 16], "job").await.unwrap(); + assert!(q.forget([1; 16], "job").await.is_err()); + q.finish(&task, Ok(Some(vec![]))).await.unwrap(); + assert_eq!( + q.status([1; 16], "job").await.unwrap().status, + Status::Cancelled + ); + q.forget([1; 16], "job").await.unwrap(); + add(&q, 1, "job", 0).await; + assert_ne!( + q.status([1; 16], "job").await.unwrap().run_id(), + task.run_id() + ); + } + #[tokio::test] + async fn retries_back_off_and_eventually_stop() { + let (q, clock) = queue(Default::default()).await; + add(&q, 1, "job", 0).await; + for attempt in 1..=MAX_ATTEMPTS { + let task = q.due().await.unwrap(); + q.begin(&task).await.unwrap(); + q.finish(&task, Err(anyhow::anyhow!("failed"))) + .await + .unwrap(); + let record = q.status([1; 16], "job").await.unwrap(); + assert_eq!(record.attempts, attempt); + assert!(q.due().await.is_none()); + clock.0.store(record.run_at, Ordering::Relaxed); + } + assert!(q.due().await.is_none()); + assert_eq!( + q.status([1; 16], "job").await.unwrap().status, + Status::Failed + ); + } + /// The recorded error names why, not only the outermost context. + #[tokio::test] + async fn a_failure_records_its_cause() { + let (q, _) = queue(Default::default()).await; + add(&q, 1, "job", 0).await; + let task = q.due().await.unwrap(); + q.begin(&task).await.unwrap(); + let error = anyhow::anyhow!("co-signer refused the run id").context("run-task failed"); + q.finish(&task, Err(error)).await.unwrap(); + let record = q.status([1; 16], "job").await.unwrap(); + assert_eq!( + record.error.as_deref(), + Some("run-task failed: co-signer refused the run id") + ); + } + /// The same backoff, but driven by a guest that really fails. + /// + /// `retries_back_off_and_eventually_stop` hands the error to `finish` + /// directly, so the component is never asked. That shape cannot show the + /// two things this one is for: that a failure *inside* the guest reaches + /// the queue at all, and that the occurrence keeps its run id across every + /// retry — the at-least-once promise, checked against a real guest rather + /// than against a synthetic error. + #[tokio::test] + #[ignore = "requires built guest-http component"] + async fn a_guest_that_returns_an_error_retries_until_the_occurrence_is_failed() { + let (q, clock) = queue(Default::default()).await; + let handle = guest(&q).await; + q.enqueue([1; 16], "job".into(), b"fail".to_vec(), 0, None) + .await + .unwrap(); + let run_id = q.status([1; 16], "job").await.unwrap().run_id(); + + for attempt in 1..=MAX_ATTEMPTS { + let task = q.due().await.unwrap(); + assert_eq!(task.run_id(), run_id, "a retry was given a new run id"); + let result = handle.run_background(&q, &task).await; + let error = format!("{:#}", result.as_ref().expect_err("the guest failed")); + assert!( + error.contains("example task failure"), + "the guest's own message did not survive the host round trip: {error}" + ); + q.finish(&task, result).await.unwrap(); + let record = q.status([1; 16], "job").await.unwrap(); + assert_eq!(record.attempts, attempt); + assert!(q.due().await.is_none()); + clock.0.store(record.run_at, Ordering::Relaxed); + } + + assert!(q.due().await.is_none()); + assert_eq!( + q.status([1; 16], "job").await.unwrap().status, + Status::Failed + ); + + // The guest writes one file per run id, and only on success. Nothing + // here ever succeeded, so nothing should be there — which is what + // separates five real invocations from one cached result replayed five + // times. The directory itself exists because the guest creates it + // before it looks at the payload. + let root = crate::tenant::existing_tenant_root(&q.fs, [1; 16]) + .await + .unwrap(); + let dir = q.fs.lookup_at(&root, "http-example").await.unwrap(); + let tasks = q.fs.lookup_at(&dir, "tasks").await.unwrap(); + let names: Vec = + q.fs.read_dir(&tasks) + .await + .unwrap() + .into_iter() + .map(|e| e.name) + .collect(); + assert!( + !names.iter().any(|n| n.starts_with("job")), + "a task that only ever failed still wrote a result: {names:?}" + ); + } + #[tokio::test] + async fn recurring_checks_coalesce_and_do_not_replay_missed_intervals() { + let (q, clock) = queue(Default::default()).await; + q.enqueue([1; 16], "job".into(), vec![], 0, Some(1000)) + .await + .unwrap(); + let first = q.due().await.unwrap(); + q.begin(&first).await.unwrap(); + clock.0.store(1_000_000, Ordering::Relaxed); + q.finish(&first, Ok(Some(vec![]))).await.unwrap(); + let next = q.status([1; 16], "job").await.unwrap(); + assert!(next.run_at > 1_000_000 && next.run_at <= 1_001_000); + assert_eq!(next.occurrence, 1); + assert!(q.due().await.is_none()); + clock.0.store(next.run_at, Ordering::Relaxed); + assert_ne!(q.due().await.unwrap().run_id(), first.run_id()); + } + #[tokio::test] + async fn unpublished_files_are_ignored_but_live_corruption_is_refused() { + let (q, _) = queue(Default::default()).await; + let h = + q.fs.open(&format!("{DIR}/orphan.tmp"), OpenFlags::create_new()) + .await + .unwrap(); + q.fs.pwrite(&h, 0, b"incomplete").await.unwrap(); + q.fs.close(&h).await.unwrap(); + let recovered = TaskQueue::open(q.fs.clone(), q.clock.clone(), Default::default()) + .await + .unwrap(); + assert!(recovered.due().await.is_none()); + let h = + q.fs.open(&format!("{DIR}/bad"), OpenFlags::create_new()) + .await + .unwrap(); + q.fs.close(&h).await.unwrap(); + assert!( + TaskQueue::open(q.fs.clone(), q.clock.clone(), Default::default()) + .await + .is_err() + ); + } + async fn guest(q: &Arc) -> Arc { + guest_with_pool(q, Arc::new(crate::Tenancy::new(Default::default()))).await + } + async fn guest_with_pool( + q: &Arc, + pool: Arc, + ) -> Arc { + let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-http/target/wasm32-wasip2/release/guest-http.wasm"); + let bytes = std::fs::read(path).expect("build examples/guest-http for wasm32-wasip2"); + let (logs, _collector) = crate::guest_io::start(Arc::new(crate::TracingLogSink)); + let env = crate::GuestEnvironment::new( + q.fs.clone(), + Box::new(HostClock), + Arc::new(nitro_nsm::fake::FakeNsm::new()), + &[], + &[], + logs, + ) + .unwrap(); + let engine = crate::ServeHandle::engine_with_watchdog().unwrap(); + let handle = crate::ServeHandle::new(&engine, &bytes, env) + .unwrap() + .with_tenancy(pool) + .with_tasks(q.clone()) + .unwrap(); + handle.verify_instantiates().await.unwrap(); + Arc::new(handle) + } + async fn request( + handle: &crate::ServeHandle, + tenant: u8, + method: &str, + path: &str, + payload: &[u8], + ) -> (u16, Vec) { + use http_body_util::BodyExt; + let body = http_body_util::Full::new(bytes::Bytes::copy_from_slice(payload)) + .map_err(|e: std::convert::Infallible| -> wasmtime_wasi_http::p2::bindings::http::types::ErrorCode { match e {} }) + .boxed_unsync(); + let req = hyper::Request::builder() + .method(method) + .header("host", "enclave.test") + .uri(path) + .body(body) + .unwrap(); + let response = handle + .handle( + wasmtime_wasi_http::p2::bindings::http::types::Scheme::Https, + req, + Some(&[tenant; 16]), + ) + .await + .unwrap(); + ( + response.status().as_u16(), + response + .into_body() + .collect() + .await + .unwrap() + .to_bytes() + .to_vec(), + ) + } + #[tokio::test] + #[ignore = "requires built guest-http component"] + async fn client_submits_then_restarted_worker_runs_only_that_tenant() { + let (q, clock) = queue(Default::default()).await; + let handle = guest(&q).await; + assert_eq!( + request(&handle, 1, "POST", "/tasks/job", b"alice").await.0, + 202 + ); + assert_eq!(request(&handle, 2, "GET", "/tasks/job", b"").await.0, 404); + drop(handle); + let reopened = TaskQueue::open(q.fs.clone(), q.clock.clone(), Default::default()) + .await + .unwrap(); + clock.0.store(20_000, Ordering::Relaxed); + let handle = guest(&reopened).await; + let task = reopened.due().await.unwrap(); + let result = handle.run_background(&reopened, &task).await; + reopened.finish(&task, result).await.unwrap(); + assert_eq!( + reopened.status([1; 16], "job").await.unwrap().status, + Status::Completed + ); + let record = reopened.status([1; 16], "job").await.unwrap(); + assert_eq!(record.result, b"alice"); + assert!(q + .fs + .lookup_at( + &crate::tenant::existing_tenant_root(&q.fs, [2; 16]) + .await + .unwrap(), + "http-example" + ) + .await + .is_err()); + assert_eq!(request(&handle, 1, "GET", "/tasks/job", b"").await.0, 200); + } + #[tokio::test] + #[ignore = "requires built guest-http component"] + async fn background_cannot_enqueue_and_cpu_loop_releases_tenant() { + let (q, _) = queue(TaskLimits { + timeout: Duration::from_millis(500), + ..Default::default() + }) + .await; + let handle = guest(&q).await; + q.enqueue( + [1; 16], + "authority".into(), + b"check-authority".to_vec(), + 0, + None, + ) + .await + .unwrap(); + let task = q.due().await.unwrap(); + let result = handle.run_background(&q, &task).await; + assert!(result.is_ok(), "{result:?}"); + q.finish(&task, result).await.unwrap(); + assert!(q.status([1; 16], "unauthorized").await.is_err()); + q.enqueue([1; 16], "loop".into(), b"spin".to_vec(), 0, None) + .await + .unwrap(); + let task = q.due().await.unwrap(); + let result = handle.run_background(&q, &task).await; + assert!(result.is_err()); + q.finish(&task, result).await.unwrap(); + assert_eq!(request(&handle, 1, "GET", "/memory", b"").await.0, 200); + } + #[tokio::test] + async fn published_task_survives_a_fresh_filesystem_mount() { + let backend = Arc::new(MemoryBackend::new()); + let master = MasterSecret::from_bytes([7; 32]); + let fs = Fs::create( + backend.clone(), + backend.clone(), + &master, + [8; 16], + Arc::new(Config::default()), + ) + .await + .unwrap(); + crate::tenant::tenant_root_by_id(&fs, [1; 16]) + .await + .unwrap(); + let clock = Arc::new(WallClockAdapter::new(Box::new(HostClock)).unwrap()); + let q = TaskQueue::open(fs.clone(), clock.clone(), Default::default()) + .await + .unwrap(); + add(&q, 1, "durable", 0).await; + let run_id = q.status([1; 16], "durable").await.unwrap().run_id(); + drop(q); + drop(fs); + let fs = Fs::mount( + backend.clone(), + backend, + &master, + [8; 16], + Arc::new(Config::default()), + None, + ) + .await + .unwrap(); + let reopened = TaskQueue::open(fs, clock, Default::default()) + .await + .unwrap(); + assert_eq!( + reopened.status([1; 16], "durable").await.unwrap().run_id(), + run_id + ); + } + + /// Recurrence, through the scheduler's own sleep and wake. + /// + /// `recurring_checks_coalesce_and_do_not_replay_missed_intervals` drives + /// `due` and `finish` by hand against a frozen clock, which is the only way + /// to ask about coalescing — and is also exactly what removes the thing + /// under test here: that the running loop re-arms itself and fires a second + /// time with nobody poking it. + /// + /// So this one takes `HostClock` deliberately. Under the tests' frozen + /// clock the loop would sleep the interval in real time and then find + /// nothing due, because `now()` never moved — and wait for ever. + #[tokio::test(flavor = "multi_thread")] + #[ignore = "requires built guest-http component"] + async fn a_recurring_task_fires_again_through_the_running_scheduler() { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([9; 32]), + [1; 16], + Arc::new(Config::default()), + ) + .await + .unwrap(); + crate::tenant::tenant_root_by_id(&fs, [1; 16]) + .await + .unwrap(); + let q = TaskQueue::open( + fs, + Arc::new(WallClockAdapter::new(Box::new(HostClock)).unwrap()), + Default::default(), + ) + .await + .unwrap(); + let handle = guest(&q).await; + + // The minimum the queue accepts, so the second occurrence is a second + // away rather than a test that waits on nothing. + q.enqueue([1; 16], "beat".into(), b"tick".to_vec(), 0, Some(1000)) + .await + .unwrap(); + let worker = tokio::spawn(q.clone().run(handle)); + + // Two files with distinct run ids, rather than `occurrence == 1`: a + // counter moving only proves a record was rewritten, where two files + // prove the guest's handler was entered a second time. + let names = tokio::time::timeout(Duration::from_secs(15), async { + loop { + assert!(!worker.is_finished(), "scheduler stopped"); + if let Ok(root) = crate::tenant::existing_tenant_root(&q.fs, [1; 16]).await { + if let Ok(dir) = q.fs.lookup_at(&root, "http-example").await { + if let Ok(tasks) = q.fs.lookup_at(&dir, "tasks").await { + let names: Vec = + q.fs.read_dir(&tasks) + .await + .unwrap() + .into_iter() + .map(|e| e.name) + .filter(|n| n.starts_with("beat")) + .collect(); + if names.len() >= 2 { + return names; + } + } + } + } + tokio::time::sleep(Duration::from_millis(50)).await; + } + }) + .await + .expect("a recurring task never fired a second time"); + + // A success resets the attempt count; a recurrence that had been + // quietly retrying would never get back to zero. But the guest writes + // its file before it returns, and the queue records the success after, + // so the second file is not yet that success: wait for the record to + // say so rather than stopping the worker between the two. + tokio::time::timeout(Duration::from_secs(15), async { + loop { + let task = q.status([1; 16], "beat").await.unwrap(); + if task.occurrence >= 2 && task.attempts == 0 { + return; + } + tokio::time::sleep(Duration::from_millis(50)).await; + } + }) + .await + .expect("the second occurrence never recorded a success"); + + worker.abort(); + let _ = worker.await; + + let unique: std::collections::BTreeSet<&String> = names.iter().collect(); + assert_eq!( + unique.len(), + names.len(), + "two occurrences shared one run id: {names:?}" + ); + } + #[tokio::test] + #[ignore = "requires built guest-http component"] + async fn busy_tenant_does_not_consume_a_retry_or_block_other_tenants() { + let (q, _) = queue(TaskLimits { + concurrency: 1, + ..Default::default() + }) + .await; + let pool = Arc::new(crate::Tenancy::new(Default::default())); + let checkout = pool.pool().checkout(&[1; 16]); + let lock = checkout.slot().tenant().clone().lock_owned().await; + let handle = guest_with_pool(&q, pool).await; + add(&q, 1, "blocked", 0).await; + add(&q, 2, "ready", 0).await; + let worker = tokio::spawn(q.clone().run(handle)); + tokio::time::timeout(Duration::from_secs(5), async { + loop { + assert!(q.records.lock().await.active.len() <= 1); + if q.status([2; 16], "ready").await.unwrap().status == Status::Completed { + break; + } + assert!(!worker.is_finished(), "scheduler stopped"); + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + assert_eq!(q.status([1; 16], "blocked").await.unwrap().attempts, 0); + assert_eq!( + q.status([1; 16], "blocked").await.unwrap().status, + Status::Pending + ); + worker.abort(); + let _ = worker.await; + drop(lock); + } + + #[tokio::test] + #[ignore = "requires built guest-http component"] + async fn missing_tenant_is_never_recreated_by_a_worker() { + let (q, _) = queue(Default::default()).await; + let handle = guest(&q).await; + add(&q, 1, "job", 0).await; + let parent = q.fs.lookup_at(&q.fs.root(), "tenants").await.unwrap(); + q.fs.rmdir(&parent, &hex::encode([1; 16])).await.unwrap(); + let task = q.due().await.unwrap(); + assert!(handle.run_background(&q, &task).await.is_err()); + assert!(crate::tenant::existing_tenant_root(&q.fs, [1; 16]) + .await + .is_err()); + } + #[tokio::test] + #[ignore = "requires built guest-http component"] + async fn simultaneous_due_tasks_stay_within_the_worker_limit() { + let (q, _) = queue(TaskLimits { + concurrency: 3, + ..Default::default() + }) + .await; + let handle = guest(&q).await; + for tenant in 1..=16 { + crate::tenant::tenant_root_by_id(&q.fs, [tenant; 16]) + .await + .unwrap(); + for job in 0..3 { + add(&q, tenant, &format!("job-{job}"), 0).await; + } + } + let worker = tokio::spawn(q.clone().run(handle)); + tokio::time::timeout(Duration::from_secs(30), async { + loop { + let r = q.records.lock().await; + assert!( + r.active.len() <= 3, + "background concurrency exceeded its limit" + ); + assert!(!worker.is_finished(), "scheduler stopped"); + if r.tasks + .values() + .all(|task| task.status == Status::Completed) + { + assert!(r + .tasks + .values() + .all(|task| task.result == vec![task.tenant[0]])); + break; + } + assert!(!r.tasks.values().any(|task| task.status == Status::Failed)); + drop(r); + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await + .expect("the due batch did not drain"); + worker.abort(); + let _ = worker.await; + } +} diff --git a/runtime/src/tenant.rs b/runtime/src/tenant.rs new file mode 100644 index 0000000..034005c --- /dev/null +++ b/runtime/src/tenant.rs @@ -0,0 +1,235 @@ +//! One filesystem, one directory per client, and a `/` that means different +//! things to different guests. +//! +//! Every client's data lives under `/tenants//` in the runtime's +//! single mounted filesystem. There is one `Fs`, one `BlockStore`, one block +//! cache and one transaction stream — a client costs a directory, not a mount. +//! +//! ## Where the separation actually is +//! +//! Not in the guest. A tenant's instance is handed its own directory as its +//! preopen, and [`s3fs_core::Fs::lookup_within`] refuses every way a path can +//! name something above it: absolute paths restart at the tenant's root rather +//! than the filesystem's, `..` stops there, and an absolute symlink target +//! resolves from there too. A guest that tries `/tenants/other/secret` gets +//! its own `tenants/other/secret`, which does not exist. +//! +//! That is a capability, not a convention, and it is why this is safe to do +//! with a shared filesystem at all. +//! +//! ## Why the identifier is derived +//! +//! `HKDF(master, sha256(client SPKI))`, so it need never be stored or looked +//! up: the same client lands on the same directory on every boot, from nothing +//! but the TLS handshake. Unguessable without the master secret, so holding a +//! client's certificate does not tell an outsider which directory is theirs. +//! +//! ## Why there is no separate register +//! +//! An earlier design gave each client their own filesystem, and then needed a +//! register to answer "should this client have one?" — because a host who hid +//! a client's root record would otherwise get a fresh, empty filesystem built +//! for them, resetting their policy and emptying their nonce ledger. +//! +//! With one filesystem that question answers itself. `/tenants/` either +//! exists in the Merkle tree or it does not, the tree is covered by one signed +//! root record, and that record is attested by the state-origin receipt at +//! boot. Hiding one client's directory means changing the root hash, which +//! fails the signature before any request is served. **The filesystem is the +//! register.** + +use std::sync::Arc; + +use anyhow::{Context, Result}; +use s3fs_core::{Fs, FsError, Inode}; + +/// The directory holding every tenant's subtree. +const TENANTS_DIR: &str = "tenants"; + +/// A client's arrival. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Arrival { + /// Their directory was already there. + Returning, + /// First contact: it was created. + New, +} + +/// A client's own directory, and how they arrived at it. +#[derive(Debug, Clone)] +pub struct TenantRoot { + pub scope: Arc, + pub tenant_id: [u8; 16], + pub arrival: Arrival, +} + +/// Find or create a client's directory in the shared filesystem. +/// +/// Cheap by construction: a lookup, and on first contact one `mkdir`. There is +/// no mount, no key derivation for a second filesystem, and no root record to +/// read — the costs the per-filesystem design would have paid on every cold +/// client. +/// Find or create a tenant's directory. +/// +/// Registration mints an id and needs the directory for it; a request resolves +/// an id from a verified assertion and needs the same. One function, so the two +/// cannot disagree about where a tenant lives. +/// +/// Cheap by construction: a lookup, and on first contact one `mkdir`. No mount, +/// no key derivation, no root record to read. +pub async fn tenant_root_by_id(fs: &Arc, tenant_id: [u8; 16]) -> Result { + let name = hex::encode(tenant_id); + let root = fs.root(); + let parent = ensure_dir(fs, &root, TENANTS_DIR).await?; + + // Looked up before it is created, so an arrival can be reported honestly: + // "new" means this runtime had never seen the tenant, which is worth a log + // line and, later, worth a policy. + let arrival = match fs.lookup_at(&parent, &name).await { + Ok(_) => Arrival::Returning, + Err(FsError::NotFound) => Arrival::New, + Err(e) => return Err(anyhow::anyhow!(e)).context("opening a tenant directory"), + }; + let scope = ensure_dir(fs, &parent, &name).await?; + Ok(TenantRoot { + scope, + tenant_id, + arrival, + }) +} + +/// Background work must never create a tenant whose directory is missing. +pub async fn existing_tenant_root(fs: &Arc, tenant_id: [u8; 16]) -> Result> { + let parent = fs.lookup_at(&fs.root(), TENANTS_DIR).await?; + fs.lookup_at(&parent, &hex::encode(tenant_id)) + .await + .context("scheduled tenant no longer exists") +} + +/// Open a directory, creating it if it is not there. +/// +/// `AlreadyExists` is success, not failure: two requests racing to first +/// contact both want the same directory, and whichever loses should find what +/// the winner made rather than report a conflict nobody caused. +pub async fn ensure_dir(fs: &Arc, parent: &Arc, name: &str) -> Result> { + match fs.lookup_at(parent, name).await { + Ok(dir) => Ok(dir), + Err(FsError::NotFound) => match fs.mkdir(parent, name).await { + Ok(dir) => Ok(dir), + Err(FsError::AlreadyExists) => fs + .lookup_at(parent, name) + .await + .map_err(|e| anyhow::anyhow!(e)) + .with_context(|| format!("opening {name} after losing the race to create it")), + Err(e) => Err(anyhow::anyhow!(e)).with_context(|| format!("creating {name}")), + }, + Err(e) => Err(anyhow::anyhow!(e)).with_context(|| format!("opening {name}")), + } +} + +/// Every client with a directory. Diagnostics, and a future sweeper. +pub async fn tenants(fs: &Arc) -> Result> { + let root = fs.root(); + let dir = match fs.lookup_at(&root, TENANTS_DIR).await { + Ok(dir) => dir, + Err(FsError::NotFound) => return Ok(Vec::new()), + Err(e) => return Err(anyhow::anyhow!(e)).context("opening the tenants directory"), + }; + Ok(fs + .read_dir(&dir) + .await + .map_err(|e| anyhow::anyhow!(e)) + .context("listing tenants")? + .into_iter() + .map(|e| e.name) + .collect()) +} + +#[cfg(test)] +mod tests { + use super::*; + use s3fs_core::backend::memory::MemoryBackend; + use s3fs_core::{Config, MasterSecret}; + + async fn shared_fs() -> Arc { + let backend = Arc::new(MemoryBackend::new()); + Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([7u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("the shared filesystem") + } + + /// A returning client lands where it left its data. If this were not + /// stable their state would look lost on every request. + #[tokio::test] + async fn a_client_returns_to_the_same_directory() { + let fs = shared_fs().await; + let first = tenant_root_by_id(&fs, [0xab; 16]).await.unwrap(); + let second = tenant_root_by_id(&fs, [0xab; 16]).await.unwrap(); + + assert_eq!(first.tenant_id, second.tenant_id); + assert_eq!(first.scope.objid(), second.scope.objid()); + assert_eq!(first.arrival, Arrival::New); + assert_eq!(second.arrival, Arrival::Returning); + } + + #[tokio::test] + async fn two_clients_get_two_directories() { + let fs = shared_fs().await; + let a = tenant_root_by_id(&fs, [0xaa; 16]).await.unwrap(); + let b = tenant_root_by_id(&fs, [0xbb; 16]).await.unwrap(); + assert_ne!(a.tenant_id, b.tenant_id); + assert_ne!(a.scope.objid(), b.scope.objid()); + } + + /// A registration and a later request must land in the same directory, or + /// a passkey would be registered against storage it could never reach. + #[tokio::test] + async fn registration_and_a_request_land_in_the_same_directory() { + let fs = shared_fs().await; + let at_registration = tenant_root_by_id(&fs, [0xcd; 16]).await.unwrap(); + let at_request = tenant_root_by_id(&fs, [0xcd; 16]).await.unwrap(); + assert_eq!(at_request.scope.objid(), at_registration.scope.objid()); + assert_eq!(at_registration.arrival, Arrival::New); + assert_eq!(at_request.arrival, Arrival::Returning); + } + + /// Losing the race to create a directory must not be an error: both + /// callers want the same directory. + #[tokio::test] + async fn ensure_dir_is_idempotent() { + let fs = shared_fs().await; + let root = fs.root(); + let first = ensure_dir(&fs, &root, "runtime").await.unwrap(); + let second = ensure_dir(&fs, &root, "runtime").await.unwrap(); + assert_eq!(first.objid(), second.objid()); + } + + #[tokio::test] + async fn tenants_are_listed_and_none_is_not_an_error() { + let fs = shared_fs().await; + assert!(tenants(&fs).await.unwrap().is_empty()); + + let a = tenant_root_by_id(&fs, [0xaa; 16]).await.unwrap(); + let b = tenant_root_by_id(&fs, [0xbb; 16]).await.unwrap(); + let listed = tenants(&fs).await.unwrap(); + assert_eq!(listed.len(), 2); + assert!(listed.contains(&hex::encode(a.tenant_id))); + assert!(listed.contains(&hex::encode(b.tenant_id))); + } + + /// Tenant directories sit under one parent, and that parent is above every + /// tenant's scope — which is what keeps `/runtime/` unreachable from a + /// guest too. + #[tokio::test] + async fn a_tenant_directory_is_not_the_filesystem_root() { + let fs = shared_fs().await; + let t = tenant_root_by_id(&fs, [0xab; 16]).await.unwrap(); + assert_ne!(t.scope.objid(), fs.root().objid()); + } +} diff --git a/runtime/src/testing/cosign.rs b/runtime/src/testing/cosign.rs new file mode 100644 index 0000000..bb164b7 --- /dev/null +++ b/runtime/src/testing/cosign.rs @@ -0,0 +1,355 @@ +//! An NSM that signs what the emulator will not. +//! +//! QEMU's `nitro-enclave` machine implements the NSM protocol but not the +//! signing. Its source says so — *"we don't actually sign the data, so we use +//! -1 as the 'alg' value"* — and -1 is not a COSE algorithm identifier. So the +//! documents it produces carry genuine contents, the real PCR0 of the image +//! that booted and the PCR16 the runtime measured its guest into, inside an +//! envelope no client can check. Every client run against the emulator has had +//! to pass `--unsigned-emulator`, which skips the signature, the certificate +//! chain and the validity windows — the part of a client most worth exercising +//! before it meets hardware, and the only part the emulator could never reach. +//! +//! This wraps the device instead of replacing it. [`CosigningNsm::attest`] asks +//! the emulator for a document, parses it, and re-signs *that same payload* +//! with a chain minted at boot. Contents stay the device's answers; only the +//! envelope becomes real. Nothing else is intercepted — entropy and every PCR +//! operation go straight through — so a client comparing PCR0 against what +//! `nix build` printed is still comparing against the hardware's own report. +//! +//! # What this does not do +//! +//! It does not make a document mean anything. The key is minted inside the same +//! image whoever runs it controls, so a document proves only that its producer +//! had that key, which is everyone who can boot the image. That is the whole +//! difference from real Nitro, where the key lives in hardware and the chain +//! goes back to a root AWS publishes. +//! +//! Two things keep that from being mistaken for the real property. The root is +//! fresh every boot, so there is no long-lived value to paste into an app and +//! forget; and a client still has to set `allow_untrusted_root` to accept a +//! non-AWS root at all, which [`nitro_attestation::verify`] reports back as +//! [`Trust::SelfSigned`](nitro_attestation::Trust::SelfSigned) rather than +//! letting it pass as the real thing. + +use std::sync::Arc; +use std::time::{Duration, SystemTime}; + +use anyhow::{Context, Result}; + +/// How long the minted chain is good for. +/// +/// A real Nitro leaf is valid for about three hours, which costs AWS nothing +/// because it mints another whenever an enclave asks. This chain is minted once +/// at boot and never again, so its window has to outlast however long the +/// enclave runs — and a development enclave stays up for as long as somebody is +/// working against it. Long enough for that, short enough that a root scraped +/// out of a log is useless next quarter. +const CHAIN_VALIDITY: Duration = Duration::from_secs(30 * 24 * 3600); + +/// Backdating, for the same reason certificate issuers backdate: the enclave's +/// clock and the verifying client's need not agree to the second, and a leaf +/// that is not valid *yet* fails exactly like one that was never valid. +const CHAIN_BACKDATE: Duration = Duration::from_secs(3600); + +/// An [`Nsm`](nitro_nsm::Nsm) that delegates everything and re-signs documents. +#[derive(Debug)] +pub struct CosigningNsm { + inner: Arc, + chain: nitro_attestation::testing::TestChain, +} + +impl CosigningNsm { + /// Wrap a device, minting the chain its documents will be signed with. + pub fn wrap(inner: Arc) -> Result { + let now = SystemTime::now(); + let chain = nitro_attestation::testing::TestChain::with_validity( + now - CHAIN_BACKDATE, + now + CHAIN_VALIDITY, + ) + .context("minting the attestation signing chain")?; + Ok(CosigningNsm { inner, chain }) + } + + /// DER of the root a client must pin to verify what this signs. + /// + /// Not a secret and not a credential: it is the public half, and publishing + /// it is the only way anything can check a document. The *private* key + /// never leaves this process, which is not a security property here — see + /// the module documentation — but does mean the root is all a client needs. + pub fn trust_root(&self) -> Vec { + self.chain.root_der().to_vec() + } +} + +impl nitro_nsm::Nsm for CosigningNsm { + fn get_random(&self, buf: &mut [u8]) -> Result<()> { + self.inner.get_random(buf) + } + + /// Ask the device, then re-sign what it said. + /// + /// The payload is the device's, field for field — module id, timestamp, + /// every locked register, and the `user_data`/`nonce`/`public_key` this + /// request asked it to bind. Only `certificate` and `cabundle` are + /// replaced, and they have to be: the signature is made with the leaf's + /// key, so a document carrying the device's placeholder certificate would + /// name a key that did not sign it and fail verification for a reason that + /// looks nothing like the cause. + fn attest(&self, request: &nitro_nsm::AttestationRequest) -> Result> { + let unsigned = self.inner.attest(request)?; + let mut document = nitro_attestation::parse(&unsigned) + .context("the device produced a document this build cannot parse")?; + document.certificate = self.chain.leaf.clone(); + document.cabundle = self.chain.cabundle.clone(); + self.chain.document_from(document) + } + + fn describe_pcr(&self, index: u16) -> Result { + self.inner.describe_pcr(index) + } + + fn extend_pcr(&self, index: u16, data: &[u8]) -> Result> { + self.inner.extend_pcr(index, data) + } + + fn lock_pcr(&self, index: u16) -> Result<()> { + self.inner.lock_pcr(index) + } + + fn describe(&self) -> String { + format!("{} (documents re-signed at boot)", self.inner.describe()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nitro_nsm::Nsm as _; + use std::collections::BTreeMap; + + /// A device shaped like QEMU's: a real payload in an envelope with `alg = + /// -1` and no signature. + /// + /// Written out rather than borrowed from `nitro_nsm::fake`, whose `attest` + /// deliberately returns a fixed byte string that is not a document at all. + /// What is under test here is what happens *to a document*, so the stand-in + /// has to produce one. + #[derive(Debug)] + struct EmulatorNsm { + inner: nitro_nsm::fake::FakeNsm, + } + + impl EmulatorNsm { + fn new() -> Self { + EmulatorNsm { + inner: nitro_nsm::fake::FakeNsm::new(), + } + } + } + + impl nitro_nsm::Nsm for EmulatorNsm { + fn get_random(&self, buf: &mut [u8]) -> Result<()> { + self.inner.get_random(buf) + } + fn describe_pcr(&self, index: u16) -> Result { + self.inner.describe_pcr(index) + } + fn extend_pcr(&self, index: u16, data: &[u8]) -> Result> { + self.inner.extend_pcr(index, data) + } + fn lock_pcr(&self, index: u16) -> Result<()> { + self.inner.lock_pcr(index) + } + fn describe(&self) -> String { + "emulated NSM".to_string() + } + + fn attest(&self, request: &nitro_nsm::AttestationRequest) -> Result> { + // Locked registers only, as a real document lists. + let mut pcrs = BTreeMap::new(); + for index in 0..32u16 { + let pcr = self.inner.describe_pcr(index)?; + if pcr.locked { + pcrs.insert(u32::from(index), pcr.value); + } + } + let payload = nitro_attestation::testing::encode_payload( + &nitro_attestation::AttestationDocument { + module_id: "i-0emulated000000000".to_string(), + timestamp_ms: 1_700_000_000_000, + digest: "SHA384".to_string(), + pcrs, + // No certificate and no chain, which is consistent of it: + // it has no key to name one for. + certificate: Vec::new(), + cabundle: Vec::new(), + public_key: request.public_key.clone(), + user_data: request.user_data.clone(), + nonce: request.nonce.clone(), + }, + ); + // A COSE_Sign1 by shape only: protected headers say `alg = -1` + // (`a1 01 20`) and the signature is empty. + let mut out = Vec::new(); + ciborium::into_writer( + &ciborium::Value::Array(vec![ + ciborium::Value::Bytes(vec![0xa1, 0x01, 0x20]), + ciborium::Value::Map(vec![]), + ciborium::Value::Bytes(payload), + ciborium::Value::Bytes(Vec::new()), + ]), + &mut out, + ) + .unwrap(); + Ok(out) + } + } + + fn wrapped() -> (Arc, CosigningNsm) { + let device = Arc::new(EmulatorNsm::new()); + let cosigning = CosigningNsm::wrap(device.clone()).expect("minting the chain"); + (device, cosigning) + } + + /// The strict check, which is the whole reason for co-signing. + /// + /// `allow_untrusted_root` stays **false**: with it on, `verify` accepts + /// whatever root the document arrived with and reports + /// [`Trust::SelfSigned`](nitro_attestation::Trust::SelfSigned), so a + /// "pinned" root pins nothing. Off, the presented root must equal this one + /// — the same comparison a client makes against AWS's. + fn options(trust_root: Vec) -> nitro_attestation::VerifyOptions { + nitro_attestation::VerifyOptions { + trust_root, + now: SystemTime::now(), + allow_untrusted_root: false, + } + } + + /// The whole point. Both halves are asserted together because either alone + /// could pass while the feature does nothing: a verifier that accepted the + /// emulator's own document would make the second half meaningless. + #[test] + fn a_document_the_emulator_would_not_sign_verifies_once_co_signed() { + let (device, cosigning) = wrapped(); + let request = nitro_nsm::AttestationRequest::with_user_data(b"cert-hash".to_vec()); + + let bare = device.attest(&request).unwrap(); + nitro_attestation::verify(&bare, &options(cosigning.trust_root())) + .expect_err("the emulator's unsigned document must not verify"); + + let signed = cosigning.attest(&request).unwrap(); + let verified = nitro_attestation::verify(&signed, &options(cosigning.trust_root())) + .expect("the co-signed document must verify"); + assert_eq!( + verified.document.user_data.as_deref(), + Some(&b"cert-hash"[..]) + ); + } + + /// Re-signing must not become re-authoring. The registers a client pins are + /// the device's answers, so they have to survive the round trip byte for + /// byte — including one the runtime extended and locked, which is how PCR16 + /// reaches a document at all. + #[test] + fn the_co_signed_document_carries_the_devices_own_registers() { + let (device, cosigning) = wrapped(); + let pcr16 = device.extend_pcr(16, b"a guest component").unwrap(); + device.lock_pcr(16).unwrap(); + + let signed = cosigning + .attest(&nitro_nsm::AttestationRequest::default()) + .unwrap(); + let verified = + nitro_attestation::verify(&signed, &options(cosigning.trust_root())).unwrap(); + + assert_eq!( + verified.document.pcr(0), + Some(&device.describe_pcr(0).unwrap().value[..]), + "PCR0 must be the device's measurement, not one this wrapper invented" + ); + assert_eq!(verified.document.pcr(16), Some(&pcr16[..])); + assert_eq!(verified.document.module_id, "i-0emulated000000000"); + } + + /// A nonce is what stops a document from an earlier boot being passed off + /// as current, so it has to be the *caller's* bytes that come back. + #[test] + fn the_nonce_the_caller_asked_to_bind_comes_back_unchanged() { + let (_device, cosigning) = wrapped(); + let request = nitro_nsm::AttestationRequest::with_user_data(b"cert-hash".to_vec()) + .nonce(b"a fresh nonce".to_vec()); + + let signed = cosigning.attest(&request).unwrap(); + let verified = + nitro_attestation::verify(&signed, &options(cosigning.trust_root())).unwrap(); + assert_eq!( + verified.document.nonce.as_deref(), + Some(&b"a fresh nonce"[..]) + ); + } + + /// Otherwise the chain check is decorative: a client that would accept a + /// document signed by anybody has gained nothing over `--unsigned-emulator`. + #[test] + fn a_client_pinning_a_different_root_refuses_it() { + let (_device, cosigning) = wrapped(); + let stranger = nitro_attestation::testing::TestChain::new().unwrap(); + + let signed = cosigning + .attest(&nitro_nsm::AttestationRequest::default()) + .unwrap(); + nitro_attestation::verify(&signed, &options(stranger.root_der().to_vec())) + .expect_err("a document must not verify against a root that did not issue it"); + } + + /// The claim worth making about the whole feature: pinning the minted root + /// is the *same* check a client runs against AWS, reported the same way — + /// so the code path under development is the production one, not a relaxed + /// variant of it. And a client that pins nothing else still refuses this: + /// the default options carry the AWS root, and they must say no. + #[test] + fn pinning_the_minted_root_is_the_check_a_client_makes_against_aws() { + let (_device, cosigning) = wrapped(); + let signed = cosigning + .attest(&nitro_nsm::AttestationRequest::default()) + .unwrap(); + + let verified = + nitro_attestation::verify(&signed, &options(cosigning.trust_root())).unwrap(); + assert_eq!( + verified.trust, + nitro_attestation::Trust::ChainVerified, + "a pinned root that matches must verify as a chain, not as self-signed" + ); + + nitro_attestation::verify(&signed, &nitro_attestation::VerifyOptions::default()) + .expect_err("nothing minted here may pass against the AWS root"); + } + + /// Everything except `attest` is the device's. Stated as a test because the + /// tempting next change — caching PCRs in the wrapper — would break the + /// property that a client comparing PCR0 against `nix build` is comparing + /// against the hardware's own report. + #[test] + fn nothing_but_the_signature_is_intercepted() { + let (device, cosigning) = wrapped(); + + assert_eq!( + cosigning.extend_pcr(16, b"x").unwrap(), + device.describe_pcr(16).unwrap().value, + "an extend through the wrapper must land on the device" + ); + cosigning.lock_pcr(16).unwrap(); + assert!(device.describe_pcr(16).unwrap().locked); + cosigning + .extend_pcr(16, b"y") + .expect_err("the device's own refusal must come through"); + + let mut buf = [0u8; 8]; + cosigning.get_random(&mut buf).unwrap(); + assert_ne!(buf, [0u8; 8], "entropy must come from the device"); + assert!(cosigning.describe().contains("emulated NSM")); + } +} diff --git a/runtime/src/testing/fcm-key.pem b/runtime/src/testing/fcm-key.pem new file mode 100644 index 0000000..db4b7ce --- /dev/null +++ b/runtime/src/testing/fcm-key.pem @@ -0,0 +1,28 @@ +-----BEGIN PRIVATE KEY----- +MIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQCvw2xq8t9LTXUH +cyeYu7ZfDpvMoXMRXOtXTMczYrU6/L6XLFWacxeSdyCeEhzHLASDaR6UIQaVv4ic +achgEJ8MCitwXf99VJ1h5IXFatdZJdGoI9FifWAfN/FKutFgPjO49FSdvJo8fG4u +Zr1hJ201LKrH4j9DvG1aSTk+5Q94f3p+ZHPJ2FkFjAicNFPprlVTm6H7Sct7VWYc +oRXkBMx4TxyhWm+AHZlFzKJ6ZIR8SLC+Ieu7+5YVsE3xvUBVMvrLcGpFl/ioMbTY +HMTFooefdYbKaelUaJciv9lZnWKXhyvYPuYZwcC9sk09jeIWi9OlaBpDosPt6K70 +NQ2gWe5VAgMBAAECggEAHD/DTOQoteRm3xHnxxFCdEA3k7nOMff2fkNBj/V5Ldgh +9M+kGY0OeJSreiRsmiltt0Y9qy6srYRJe2Q4F5KMUYXP6gE9l0HygqGVS3/KyVH+ +ArFxDYyblqDp5+ojTT3qF7uzXt/JhVe1aMFMBlGtKHr7nuEy7FrcU4LJz91GcYX9 +Guj2oJJNwvBmoDR2w+4iJ7UZeYvpuqJ1qIs/YZVaE7dzFgyIbA5AR2s2/F+/OVDd +ePsqPDLOFHEyyhfYswMYaHaCMhixZkx6+y1i++lx0GIb2Thjb61TBWu/jHDsTB43 +twI/V2NIcmYQmUrXODPwbW/+kFBlz+Cf5863dLzAKQKBgQDq0Lq/BmRzLY+u3UDo +fYd27bz1rJif2DEmXonAWrfxTBJIoQrFKX9SsyzPJPbnKACuHB4jTVETHhjccKJ3 +FGHz1k986NaGo6iJ5UBin+Td3rIOzNrp781PZRBsLJW8P/siO14EZzFKlssFdkva +ZMnCkQlWOBb2/42zLvqMwxZaPQKBgQC/ntX4BiBWtJ0I6hHtjHEwMmZh0wMFBF8G +8zpgUCySClnjaoK6kLA9hwy4zDPDOtIZOieHYs+Rg16EjsEMeCS52gmrpdpEE8ob +eCKQNViEPOXm3pJBsuIlVrodaWO4Oo1Pvhc8LZBTZ63qvXV0/pKzfnWAjS88Vwc0 +8nxUlWRd+QKBgB3tpq+sP+dSOkr+VkSLo1VsLbZeXkGZS4JpcEM9DM7LdFUfeYDx +rhG7Vo28V1/VAGkwmkLDmv7FykNmc76bsXRjr1PrVVRpzZRtzMwFNyV0OdubDpfc +gZ2J8xLmh9sriHWvfWcwQ98O4yd6EWbvi6up0rfThFHM9qGM7lA8mT+9AoGAZHUr +C8p6bbpmkWPVXkpAlNn3XtW3QYwXHZeqRRADLdULZvRR8Okl3DvO6Zr0kCdoOh2I +16tv0oOiq7ADeTwLVPwAEeLzWLlfPaNvy1aMP1eF19Fbr+HOOXEMRZsY0l6v8txf +ZgclIPS78tK8n0dPNZbYlzptRx8BAjsV/2oKolECgYA0sPm2WbpwEApgeMc+nXbP +tbmPLZcEpXSm9Z3UdM0GnkY3DBGGRXyH1At3KlRIgBq4zN8ojNKn8cbSgKJ/a8aT +Ly6fCf18shOUq3jww1TZOa1TkUZFWNfdTjTIMXEzH/dvIlJtlJjWmStp+qrlBKYE +GBs4U6iXoioFnUaCwPT65w== +-----END PRIVATE KEY----- diff --git a/runtime/src/testing/mod.rs b/runtime/src/testing/mod.rs new file mode 100644 index 0000000..f8d924a --- /dev/null +++ b/runtime/src/testing/mod.rs @@ -0,0 +1,515 @@ +//! A whole enclave, in process, for people writing guests. +//! +//! The runtime's own integration tests each stood up their own listener, their +//! own signing NSM and their own hand-rolled HTTPS client. That was tolerable +//! while the only guests lived in this repository. It is not tolerable as the +//! way a *separate* project — one whose component this runtime will actually +//! serve — finds out whether its guest works, because it would have to +//! reimplement all of it before writing a single assertion. +//! +//! So this is that harness, exported: +//! +//! ```no_run +//! # async fn f() -> anyhow::Result<()> { +//! use enclave_runtime::testing::Enclave; +//! +//! let enclave = Enclave::builder(std::fs::read("cosigner.wasm")?) +//! .background_tasks() +//! .notify() +//! .start() +//! .await?; +//! +//! let alice = enclave.enrol().await?; +//! let (status, body) = enclave.signed(&alice, "POST", "/sign", "payload").await?; +//! # Ok(()) } +//! ``` +//! +//! Everything below the guest is real: a real TLS handshake, a real WebAuthn +//! ceremony, real interaction tokens scoped to one route, real tenant +//! isolation, the real task scheduler, and attestation documents that verify +//! against a chain this harness mints. +//! +//! # What it is not +//! +//! **The NSM is a test chain, not hardware.** Documents from here verify only +//! against [`Enclave::trust_root`], which nothing outside this process has any +//! reason to trust. A client that accepts them under production settings has a +//! bug this harness cannot find. +//! +//! **PCR0 is a constant**, not a measurement of a real image, and the master +//! key is static rather than released by KMS against an attestation. Those two +//! are the whole of what an enclave buys, and neither is exercised here — the +//! QEMU harness in `deploy/qemu-nitro/` covers as much of them as an emulator +//! can, and real hardware covers the rest. +//! +//! Behind the `testing` feature, like [`crate::SoftwareAuthenticator`], because +//! it mints assertions and signs attestations. Nothing here may reach an image. + +use std::net::SocketAddr; +use std::sync::Arc; + +use anyhow::{Context, Result}; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +use crate::auth::SoftwareAuthenticator; + +mod cosign; +mod nsm; +mod stub; + +pub use cosign::CosigningNsm; +pub use nsm::SigningNsm; + +/// The relying party this harness serves as. +pub const RP_ID: &str = "enclave.test"; +/// The origin its assertions claim. +pub const ORIGIN: &str = "https://enclave.test"; + +/// What `SigningNsm` reports as PCR0. A constant, and honest about it: nothing +/// measured this. +pub const PCR0: [u8; 48] = [0x5a; 48]; + +/// One wake signal the guest raised, as it reached the wire. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Wake { + pub category: String, + pub reference: Option, + pub token: String, +} + +/// A registered passkey, ready to sign for requests. +pub struct Credential { + authenticator: SoftwareAuthenticator, + tenant: String, +} + +impl Credential { + /// The tenant this passkey resolved to, hex encoded. + pub fn tenant(&self) -> &str { + &self.tenant + } + + pub fn credential_id(&self) -> &[u8] { + self.authenticator.credential_id() + } +} + +/// How to build one. +pub struct EnclaveBuilder { + guest: Vec, + background_tasks: bool, + notify: bool, +} + +impl EnclaveBuilder { + /// Let the guest schedule durable work. Requires a `run-task` export. + pub fn background_tasks(mut self) -> Self { + self.background_tasks = true; + self + } + + /// Let the guest enrol devices and raise wake signals. + /// + /// The harness stands up a stub in place of Firebase and records what the + /// runtime sent — see [`Enclave::wakes`]. The transport, the signed OAuth + /// assertion and the error handling are the real ones; only the far end is + /// not Google. + pub fn notify(mut self) -> Self { + self.notify = true; + self + } + + pub async fn start(self) -> Result { + Enclave::start(self).await + } +} + +/// A running enclave. +pub struct Enclave { + addr: SocketAddr, + nsm: Arc, + guest: Vec, + fcm: Option>, +} + +impl Enclave { + pub fn builder(guest: impl Into>) -> EnclaveBuilder { + EnclaveBuilder { + guest: guest.into(), + background_tasks: false, + notify: false, + } + } + + async fn start(builder: EnclaveBuilder) -> Result { + use crate::{ + AuthEndpoints, ChallengeStore, FilesystemCredentials, Gate, GuestEnvironment, + HostClock, PoolLimits, ServeConfig, Tenancy, TlsIdentity, + }; + use s3fs_core::backend::memory::MemoryBackend; + use s3fs_core::{Config, Fs, MasterSecret}; + + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([9u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .context("creating the harness filesystem")?; + + let nsm = Arc::new(SigningNsm::new()?); + // As `main` does, before anything could attest: the documents this + // enclave produces carry PCR16 because the guest was measured into it. + crate::measure_guest(nsm.as_ref(), &builder.guest).context("measuring the guest")?; + let entropy: Arc = nsm.clone(); + + let credentials = Arc::new(FilesystemCredentials::new(fs.clone())); + let gate = Arc::new(Gate::new( + crate::build_relying_party(RP_ID, ORIGIN, &[])?, + ChallengeStore::new(std::time::Duration::from_secs(60), 256), + credentials.clone(), + crate::TokenStore::new( + std::time::Duration::from_secs(60), + crate::DEFAULT_TOKEN_CAPACITY, + ), + )); + let auth = Arc::new(AuthEndpoints::new( + gate.clone(), + credentials, + fs.clone(), + entropy.clone(), + )); + + // Detached: the collector runs as long as this environment can send, + // which is what a harness wants. Production drains it explicitly. + let (logs, _collector) = crate::guest_io::start(Arc::new(crate::TracingLogSink)); + let guest_env = GuestEnvironment::new(fs, Box::new(HostClock), entropy, &[], &[], logs) + .context("building the guest environment")?; + + let fcm = if builder.notify { + Some(stub::Fcm::start().await?) + } else { + None + }; + let notify = match &fcm { + Some(stub) => Some(stub.config()?), + None => None, + }; + + let tls = TlsIdentity::self_signed(&[RP_ID.to_string()])?; + let listener = std::net::TcpListener::bind("127.0.0.1:0")?; + let addr = listener.local_addr()?; + drop(listener); + + let bytes = builder.guest.clone(); + let attestation: Arc = nsm.clone(); + tokio::spawn(async move { + let _ = crate::serve_component( + &bytes, + guest_env, + ServeConfig { + notify, + background_tasks: builder.background_tasks.then(Default::default), + addr, + certificate: Some(crate::CertificateSlot::fixed(Arc::new(tls))), + acme: None, + attestation: Some(attestation), + request_timeout: std::time::Duration::from_secs(30), + max_interaction: std::time::Duration::from_secs(300), + tenancy: Some(Arc::new(Tenancy::new(PoolLimits::default()))), + authentication: Some((auth, gate)), + egress: Default::default(), + }, + ) + .await; + }); + + for _ in 0..400 { + if tokio::net::TcpStream::connect(addr).await.is_ok() { + return Ok(Enclave { + addr, + nsm, + guest: builder.guest, + fcm, + }); + } + tokio::time::sleep(std::time::Duration::from_millis(25)).await; + } + anyhow::bail!("the harness never came up on {addr}") + } + + pub fn addr(&self) -> SocketAddr { + self.addr + } + + pub fn url(&self) -> String { + format!("https://127.0.0.1:{}", self.addr.port()) + } + + /// The root a client must trust to verify this enclave's documents. + /// + /// Nothing outside this process has any reason to, which is the point: a + /// client that verifies against its production roots will refuse these, and + /// should. + pub fn trust_root(&self) -> Vec { + self.nsm.trust_root() + } + + pub fn pcr0(&self) -> [u8; 48] { + PCR0 + } + + /// What the guest measured to, as a key policy would pin it. + pub fn pcr16(&self) -> [u8; 48] { + nitro_attestation::guest_pcr(&self.guest) + } + + /// Every wake signal the guest raised, in order. + /// + /// Empty unless the builder asked for [`EnclaveBuilder::notify`]. + pub fn wakes(&self) -> Vec { + self.fcm.as_ref().map(|f| f.wakes()).unwrap_or_default() + } + + /// Register a passkey, and with it a new tenant. + pub async fn enrol(&self) -> Result { + let authenticator = SoftwareAuthenticator::new(RP_ID); + let (status, options) = self + .post_json("/auth/register/options", serde_json::json!({})) + .await?; + anyhow::ensure!(status == 200, "registration was refused: {options}"); + let challenge = options["options"]["publicKey"]["challenge"] + .as_str() + .context("registration options carry no challenge")?; + let (status, body) = self + .post_json( + "/auth/register/verify", + serde_json::json!({ + "registration_id": options["registration_id"], + "credential": authenticator.register(challenge, ORIGIN), + }), + ) + .await?; + anyhow::ensure!(status == 200, "registration was refused: {body}"); + Ok(Credential { + tenant: body["tenant_id"] + .as_str() + .context("no tenant id")? + .to_string(), + authenticator, + }) + } + + /// A signed request: ask for a challenge bound to it, sign, then send it. + /// + /// Three round trips, because a passkey is a challenge and a response and + /// the approval names this exact method and path. + pub async fn signed( + &self, + credential: &Credential, + method: &str, + path: &str, + body: &str, + ) -> Result<(u16, String)> { + let token = self.token_for(credential, method, path).await?; + let (status, _, body) = self + .request( + method, + path, + &[("authorization", format!("Bearer {token}"))], + body, + ) + .await?; + Ok((status, body)) + } + + async fn token_for(&self, credential: &Credential, method: &str, path: &str) -> Result { + use base64::Engine as _; + let b64 = base64::engine::general_purpose::URL_SAFE_NO_PAD; + let (path_only, query) = match path.split_once('?') { + Some((p, q)) => (p, Some(q)), + None => (path, None), + }; + let (status, options) = self + .post_json( + "/auth/request/options", + serde_json::json!({ + "credential_id": b64.encode(credential.credential_id()), + "method": method, + "path": path_only, + "query": query, + }), + ) + .await?; + anyhow::ensure!(status == 200, "no challenge was issued: {options}"); + + let challenge = options["options"]["publicKey"]["challenge"] + .as_str() + .context("challenge options carry no challenge")?; + let assertion = credential.authenticator.assert(challenge, ORIGIN); + let (status, body) = self + .post_json( + "/auth/request/verify", + serde_json::json!({ + "challenge_id": options["challenge_id"], + "assertion": b64.encode(assertion.to_string().as_bytes()), + }), + ) + .await?; + anyhow::ensure!(status == 200, "the assertion was refused: {body}"); + body["token"] + .as_str() + .map(str::to_string) + .context("no token in the response") + } + + async fn post_json( + &self, + path: &str, + body: serde_json::Value, + ) -> Result<(u16, serde_json::Value)> { + let (status, _, text) = self + .request( + "POST", + path, + &[("content-type", "application/json".to_string())], + &body.to_string(), + ) + .await?; + Ok(( + status, + serde_json::from_str(&text).unwrap_or(serde_json::Value::Null), + )) + } + + /// One HTTPS request, returning status, headers and body. + /// + /// The certificate is accepted without checking, and that is correct here: + /// nothing vouches for a self-signed harness certificate, and what a real + /// client checks instead is that the attestation document names the + /// certificate it was served — which [`Enclave::trust_root`] makes possible. + pub async fn request( + &self, + method: &str, + path: &str, + headers: &[(&str, String)], + body: &str, + ) -> Result<(u16, String, String)> { + use base64::Engine as _; + + let config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions()? + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAny)) + .with_no_client_auth(); + let connector = tokio_rustls::TlsConnector::from(Arc::new(config)); + let name = rustls::pki_types::ServerName::try_from(RP_ID)?; + let socket = tokio::net::TcpStream::connect(self.addr).await?; + let mut stream = connector.connect(name, socket).await?; + + let mut nonce = vec![0u8; 20]; + getrandom::fill(&mut nonce).map_err(|e| anyhow::anyhow!("nonce: {e}"))?; + let mut request = format!( + "{method} {path} HTTP/1.1\r\nHost: {RP_ID}\r\nConnection: close\r\n\ + x-enclave-nonce: {}\r\nContent-Length: {}\r\n", + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(&nonce), + body.len() + ); + for (name, value) in headers { + request.push_str(&format!("{name}: {value}\r\n")); + } + request.push_str("\r\n"); + request.push_str(body); + stream.write_all(request.as_bytes()).await?; + + let mut raw = Vec::new(); + let _ = stream.read_to_end(&mut raw).await; + let split = raw + .windows(4) + .position(|w| w == b"\r\n\r\n") + .context("no header terminator in the response")?; + let head = String::from_utf8_lossy(&raw[..split]).to_string(); + let status: u16 = head + .lines() + .next() + .and_then(|l| l.split_whitespace().nth(1)) + .context("no status line")? + .parse()?; + let rest = &raw[split + 4..]; + let body = if head + .to_ascii_lowercase() + .contains("transfer-encoding: chunked") + { + dechunk(rest) + } else { + rest.to_vec() + }; + Ok((status, head, String::from_utf8_lossy(&body).to_string())) + } +} + +fn dechunk(body: &[u8]) -> Vec { + let mut out = Vec::new(); + let mut rest = body; + while let Some(end) = rest.windows(2).position(|w| w == b"\r\n") { + let Ok(size) = usize::from_str_radix(String::from_utf8_lossy(&rest[..end]).trim(), 16) + else { + break; + }; + if size == 0 { + break; + } + let start = end + 2; + if start + size > rest.len() { + break; + } + out.extend_from_slice(&rest[start..start + size]); + rest = &rest[(start + size + 2).min(rest.len())..]; + } + out +} + +#[derive(Debug)] +struct AcceptAny; + +impl rustls::client::danger::ServerCertVerifier for AcceptAny { + fn verify_server_cert( + &self, + _e: &rustls::pki_types::CertificateDer<'_>, + _i: &[rustls::pki_types::CertificateDer<'_>], + _s: &rustls::pki_types::ServerName<'_>, + _o: &[u8], + _n: rustls::pki_types::UnixTime, + ) -> std::result::Result { + Ok(rustls::client::danger::ServerCertVerified::assertion()) + } + + fn verify_tls12_signature( + &self, + _m: &[u8], + _c: &rustls::pki_types::CertificateDer<'_>, + _d: &rustls::DigitallySignedStruct, + ) -> std::result::Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + + fn verify_tls13_signature( + &self, + _m: &[u8], + _c: &rustls::pki_types::CertificateDer<'_>, + _d: &rustls::DigitallySignedStruct, + ) -> std::result::Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + + fn supported_verify_schemes(&self) -> Vec { + rustls::crypto::aws_lc_rs::default_provider() + .signature_verification_algorithms + .supported_schemes() + } +} diff --git a/runtime/src/testing/nsm.rs b/runtime/src/testing/nsm.rs new file mode 100644 index 0000000..cedf10d --- /dev/null +++ b/runtime/src/testing/nsm.rs @@ -0,0 +1,113 @@ +//! An NSM that signs for real, echoing back whatever it was asked to bind. +//! +//! Copied into four test files before it lived here. A fake returning a canned +//! document would let an endpoint pass while binding the wrong certificate — +//! precisely the bug those tests exist to catch — so this builds the payload +//! from the request and signs it with a chain it mints. + +use anyhow::Result; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Mutex; + +/// See [`super::PCR0`]. +use super::PCR0; + +#[derive(Debug)] +pub struct SigningNsm { + chain: nitro_attestation::testing::TestChain, + pcrs: Mutex>, + /// Counted, so successive draws differ. A device that returned the same + /// bytes every time would mint one tenant id for every passkey, which is + /// the isolation property quietly inverted. + draws: AtomicU64, +} + +impl SigningNsm { + pub fn new() -> Result { + Ok(SigningNsm { + chain: nitro_attestation::testing::TestChain::new()?, + // Registers behave like the device's: 0-15 locked from the start, + // 16 free until a guest is measured into it, and documents list + // only the locked ones. So PCR16 appears in a document here for the + // reason it appears in a real one. + pcrs: Mutex::new( + (0..32) + .map(|i| nitro_nsm::Pcr { + locked: i < 16, + value: if i == 0 { + PCR0.to_vec() + } else { + nitro_nsm::PCR_ZERO.to_vec() + }, + }) + .collect(), + ), + draws: AtomicU64::new(0), + }) + } + + /// The root a client must trust to verify what this signs. + pub fn trust_root(&self) -> Vec { + self.chain.root_der().to_vec() + } +} + +impl nitro_nsm::Nsm for SigningNsm { + fn get_random(&self, buf: &mut [u8]) -> Result<()> { + let draw = self.draws.fetch_add(1, Ordering::SeqCst); + for (i, b) in buf.iter_mut().enumerate() { + *b = (i as u8) + .wrapping_mul(7) + .wrapping_add(3) + .wrapping_add(draw as u8); + } + Ok(()) + } + + fn attest(&self, request: &nitro_nsm::AttestationRequest) -> Result> { + let pcrs = self + .pcrs + .lock() + .unwrap() + .iter() + .enumerate() + .filter(|(_, pcr)| pcr.locked) + .map(|(index, pcr)| (index as u32, pcr.value.clone())) + .collect(); + self.chain + .document_with_pcrs(request.user_data.clone(), request.nonce.clone(), pcrs) + } + + fn describe_pcr(&self, index: u16) -> Result { + self.pcrs + .lock() + .unwrap() + .get(index as usize) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no PCR{index}")) + } + + fn extend_pcr(&self, index: u16, data: &[u8]) -> Result> { + let mut pcrs = self.pcrs.lock().unwrap(); + let pcr = pcrs + .get_mut(index as usize) + .ok_or_else(|| anyhow::anyhow!("no PCR{index}"))?; + anyhow::ensure!(!pcr.locked, "PCR{index} is read-only"); + pcr.value = nitro_nsm::pcr_extend(&pcr.value, data); + Ok(pcr.value.clone()) + } + + fn lock_pcr(&self, index: u16) -> Result<()> { + let mut pcrs = self.pcrs.lock().unwrap(); + let pcr = pcrs + .get_mut(index as usize) + .ok_or_else(|| anyhow::anyhow!("no PCR{index}"))?; + anyhow::ensure!(!pcr.locked, "PCR{index} is read-only"); + pcr.locked = true; + Ok(()) + } + + fn describe(&self) -> String { + "signing test NSM".into() + } +} diff --git a/runtime/src/testing/stub.rs b/runtime/src/testing/stub.rs new file mode 100644 index 0000000..352dd50 --- /dev/null +++ b/runtime/src/testing/stub.rs @@ -0,0 +1,144 @@ +//! Firebase, as far as the harness is concerned. +//! +//! The runtime builds its own FCM transport inside `serve_component`, so a +//! harness cannot hand it a recording double without putting a testing-only +//! field on `ServeConfig`. Standing up a real endpoint instead is both less +//! invasive and a better test: the transport, the signed OAuth assertion, the +//! bearer header and the error classification are all the production ones. Only +//! the far end is not Google. + +use std::net::SocketAddr; +use std::sync::{Arc, Mutex}; + +use anyhow::{Context, Result}; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +use super::Wake; +use crate::notify::{NotifyConfig, ServiceAccount}; + +const KEY: &str = include_str!("fcm-key.pem"); + +pub(super) struct Fcm { + addr: SocketAddr, + wakes: Arc>>, +} + +impl Fcm { + pub(super) async fn start() -> Result> { + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await?; + let addr = listener.local_addr()?; + let wakes = Arc::new(Mutex::new(Vec::new())); + + let recorded = wakes.clone(); + tokio::spawn(async move { + loop { + let Ok((mut socket, _)) = listener.accept().await else { + return; + }; + let recorded = recorded.clone(); + tokio::spawn(async move { + let _ = serve_one(&mut socket, &recorded).await; + }); + } + }); + + Ok(Arc::new(Fcm { addr, wakes })) + } + + pub(super) fn config(&self) -> Result { + let account = serde_json::json!({ + "type": "service_account", + "project_id": "harness", + "private_key_id": "harness", + "private_key": KEY, + "client_email": "wake@harness.iam.gserviceaccount.com", + // The assertion's `aud` is this same string, so it has to name + // where the exchange actually happens. + "token_uri": format!("http://{}/token", self.addr), + }) + .to_string(); + Ok(NotifyConfig { + project_id: "harness".into(), + service_account: ServiceAccount::parse(&account) + .context("the harness service account")?, + // `http://` is what tells the runtime plaintext is acceptable here. + endpoint: Some(format!("http://{}", self.addr)), + }) + } + + pub(super) fn wakes(&self) -> Vec { + self.wakes.lock().unwrap().clone() + } +} + +async fn serve_one(socket: &mut tokio::net::TcpStream, wakes: &Mutex>) -> Result<()> { + let mut raw = Vec::new(); + let mut buf = [0u8; 4096]; + // Read until the headers are complete, then until content-length is met. + let (head_end, length) = loop { + let read = socket.read(&mut buf).await?; + if read == 0 { + anyhow::bail!("closed before a request arrived"); + } + raw.extend_from_slice(&buf[..read]); + if let Some(end) = raw.windows(4).position(|w| w == b"\r\n\r\n") { + let head = String::from_utf8_lossy(&raw[..end]).to_ascii_lowercase(); + let length = head + .lines() + .find_map(|l| l.strip_prefix("content-length:")) + .and_then(|v| v.trim().parse::().ok()) + .unwrap_or(0); + break (end + 4, length); + } + }; + while raw.len() < head_end + length { + let read = socket.read(&mut buf).await?; + if read == 0 { + break; + } + raw.extend_from_slice(&buf[..read]); + } + + let head = String::from_utf8_lossy(&raw[..head_end]).to_string(); + let path = head + .lines() + .next() + .and_then(|l| l.split_whitespace().nth(1)) + .unwrap_or("") + .to_string(); + let body = &raw[head_end..]; + + let reply = if path.ends_with("/token") { + // The runtime signed a real RS256 assertion to get here. Verifying it + // would mean reimplementing Google; that the signing is genuine is + // covered by `notify::oauth`'s own tests. + serde_json::json!({"access_token": "harness", "expires_in": 3600}) + } else if path.ends_with("messages:send") { + let message: serde_json::Value = serde_json::from_slice(body).unwrap_or_default(); + let data = &message["message"]["data"]; + wakes.lock().unwrap().push(Wake { + category: data["category"].as_str().unwrap_or_default().to_string(), + reference: data["ref"].as_str().map(str::to_string), + token: message["message"]["token"] + .as_str() + .unwrap_or_default() + .to_string(), + }); + serde_json::json!({"name": "projects/harness/messages/stub"}) + } else { + serde_json::json!({"error": {"status": "NOT_FOUND"}}) + }; + + let encoded = reply.to_string(); + socket + .write_all( + format!( + "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\n\ + content-length: {}\r\nconnection: close\r\n\r\n{encoded}", + encoded.len() + ) + .as_bytes(), + ) + .await?; + Ok(()) +} diff --git a/crates/s3fs-wasmtime/src/bindings.rs b/runtime/src/wasi/bindings.rs similarity index 72% rename from crates/s3fs-wasmtime/src/bindings.rs rename to runtime/src/wasi/bindings.rs index d4a7a45..e566b59 100644 --- a/crates/s3fs-wasmtime/src/bindings.rs +++ b/runtime/src/wasi/bindings.rs @@ -7,18 +7,18 @@ //! and `wasi:filesystem/types/directory-entry-stream`. wasmtime::component::bindgen!({ - path: "../../wit", - world: "s3fs-host", + path: "../wit", + world: "enclave-runtime", imports: { default: async | trappable }, trappable_error_type: { - "wasi:filesystem/types.error-code" => crate::error_map::S3WasiFsError, + "wasi:filesystem/types.error-code" => crate::wasi::error_map::S3WasiFsError, }, with: { "wasi:io/poll": wasmtime_wasi::p2::bindings::io::poll, "wasi:io/streams": wasmtime_wasi::p2::bindings::io::streams, "wasi:io/error": wasmtime_wasi::p2::bindings::io::error, "wasi:clocks/wall-clock": wasmtime_wasi::p2::bindings::clocks::wall_clock, - "wasi:filesystem/types.descriptor": crate::descriptors::Descriptor, - "wasi:filesystem/types.directory-entry-stream": crate::descriptors::DirectoryEntryStream, + "wasi:filesystem/types.descriptor": crate::wasi::descriptors::Descriptor, + "wasi:filesystem/types.directory-entry-stream": crate::wasi::descriptors::DirectoryEntryStream, }, }); diff --git a/runtime/src/wasi/descriptors.rs b/runtime/src/wasi/descriptors.rs new file mode 100644 index 0000000..abb5aed --- /dev/null +++ b/runtime/src/wasi/descriptors.rs @@ -0,0 +1,62 @@ +//! Resource types stored in the wasmtime `ResourceTable`. + +use std::sync::Arc; + +use s3fs_core::{FileHandle, Inode}; + +/// Wasmtime resource handle for `wasi:filesystem/types/descriptor`. +/// +/// A descriptor is one of two flavours: +/// - **File descriptor**: an `Arc` plus the directory it was +/// opened under, so paths used in `*-at` operations resolve relative to the +/// right place. +/// - **Directory descriptor**: the directory inode itself. Reads and writes +/// against one return `is-directory`. +/// +/// The parent is captured at open time rather than walked back to on demand: +/// `at_base` is synchronous, and the parent link now lives in the dnode, which +/// would make resolving it an I/O operation. +#[derive(Debug)] +pub enum Descriptor { + File { + handle: Arc, + parent: Arc, + }, + Dir { + inode: Arc, + }, +} + +impl Descriptor { + /// The inode this descriptor opens or is rooted under. For files, the + /// file's own inode; for dirs, the directory inode itself. + pub fn inode(&self) -> &Arc { + match self { + Descriptor::File { handle, .. } => &handle.inode, + Descriptor::Dir { inode } => inode, + } + } + + /// The directory to use as the base for `*-at` path resolution. + pub fn at_base(&self) -> Arc { + match self { + Descriptor::Dir { inode } => inode.clone(), + Descriptor::File { parent, .. } => parent.clone(), + } + } +} + +/// Wasmtime resource handle for `wasi:filesystem/types/directory-entry-stream`. +/// +/// Holds a snapshot of directory entries plus an iteration cursor. +#[derive(Debug)] +pub struct DirectoryEntryStream { + pub entries: Vec, + pub cursor: usize, +} + +impl DirectoryEntryStream { + pub fn new(entries: Vec) -> Self { + Self { entries, cursor: 0 } + } +} diff --git a/crates/s3fs-wasmtime/src/error_map.rs b/runtime/src/wasi/error_map.rs similarity index 76% rename from crates/s3fs-wasmtime/src/error_map.rs rename to runtime/src/wasi/error_map.rs index 2cc059f..d8646db 100644 --- a/crates/s3fs-wasmtime/src/error_map.rs +++ b/runtime/src/wasi/error_map.rs @@ -8,7 +8,7 @@ use s3fs_core::FsError; use wasmtime_wasi::TrappableError; -use crate::bindings::wasi::filesystem::types::ErrorCode; +use crate::wasi::bindings::wasi::filesystem::types::ErrorCode; /// The host-side error type the bindgen-generated traits return. pub type S3WasiFsError = TrappableError; @@ -36,11 +36,20 @@ pub fn from_fs(e: FsError) -> ErrorCode { FsError::Loop => ErrorCode::Loop, FsError::CrossDevice => ErrorCode::CrossDevice, FsError::BadDescriptor => ErrorCode::BadDescriptor, + // A guest can never see this: it is raised while deciding whether to + // mount, long before any descriptor exists. Mapped rather than left to + // the catch-all so that adding a variant to `FsError` keeps producing + // a compile error here, which is how this arm was noticed. + FsError::NoFilesystem => ErrorCode::NoEntry, FsError::WouldBlock => ErrorCode::WouldBlock, // PreconditionFailed (CAS conflict) — surface as Exist for `O_EXCL`- // style failures; callers that want different semantics can re-map // before this point. FsError::Conflict => ErrorCode::Exist, + // Verification failures. WASI has no "the storage lied to us" code, so + // these surface as `Io` — the guest sees an unreadable filesystem + // rather than plausible-looking wrong bytes, which is the whole point. + FsError::Integrity(_) | FsError::Rollback { .. } => ErrorCode::Io, FsError::IoTimeout => ErrorCode::Io, FsError::Io(_) | FsError::Network(_) => ErrorCode::Io, } diff --git a/crates/s3fs-wasmtime/src/host_filesystem.rs b/runtime/src/wasi/host_filesystem.rs similarity index 73% rename from crates/s3fs-wasmtime/src/host_filesystem.rs rename to runtime/src/wasi/host_filesystem.rs index 98fe6f6..b218da8 100644 --- a/crates/s3fs-wasmtime/src/host_filesystem.rs +++ b/runtime/src/wasi/host_filesystem.rs @@ -7,20 +7,20 @@ //! them through `wasmtime-wasi`'s stream machinery is reserved for a follow- //! up. -use s3fs_core::{InodeKind, OpenFlags}; +use s3fs_core::{Attrs, Inode, InodeKind, OpenFlags}; use wasmtime::component::Resource; use wasmtime::Result; use wasmtime_wasi::p2::bindings::io::streams::{InputStream, OutputStream}; -use crate::bindings::wasi::filesystem::types::{ +use crate::wasi::bindings::wasi::filesystem::types::{ self as wit, Descriptor as WitDescriptor, DescriptorFlags, DescriptorStat, DescriptorType, DirectoryEntry, DirectoryEntryStream as WitDirectoryEntryStream, ErrorCode, Filesize, HostDescriptor, HostDirectoryEntryStream, MetadataHashValue, NewTimestamp, OpenFlags as WitOpenFlags, PathFlags, }; -use crate::descriptors::{Descriptor, DirectoryEntryStream}; -use crate::error_map::{from_fs, IntoS3WasiResult, S3WasiFsError, S3WasiFsResult}; -use crate::view::S3FsCtxView; +use crate::wasi::descriptors::{Descriptor, DirectoryEntryStream}; +use crate::wasi::error_map::{from_fs, IntoS3WasiResult, S3WasiFsError, S3WasiFsResult}; +use crate::wasi::view::S3FsCtxView; // --------------------------------------------------------------------------- // types::Host @@ -51,8 +51,9 @@ fn get_descriptor_owned( // Clone via match — both variants are cheap (Arc clones). let d = table.get(fd).map_err(wasmtime::Error::from)?; Ok(match d { - Descriptor::File { handle } => Descriptor::File { + Descriptor::File { handle, parent } => Descriptor::File { handle: handle.clone(), + parent: parent.clone(), }, Descriptor::Dir { inode } => Descriptor::Dir { inode: inode.clone(), @@ -60,24 +61,25 @@ fn get_descriptor_owned( }) } -fn descriptor_type_for(kind: &InodeKind) -> DescriptorType { +fn descriptor_type_for(kind: InodeKind) -> DescriptorType { match kind { InodeKind::RegularFile => DescriptorType::RegularFile, - InodeKind::Directory { .. } => DescriptorType::Directory, - InodeKind::Symlink { .. } => DescriptorType::SymbolicLink, + InodeKind::Directory => DescriptorType::Directory, + InodeKind::Symlink => DescriptorType::SymbolicLink, } } -fn stat_from_attrs(attrs: &s3fs_core::inode::Attrs, kind: &InodeKind) -> DescriptorStat { - let dt = systemtime_to_wit(attrs.last_modified); +/// Every field here is now read straight from the dnode. The link count used +/// to be hardcoded to 1 and the three timestamps were all the same value, +/// because the previous engine had nowhere to keep them. +fn stat_from_attrs(attrs: &Attrs) -> DescriptorStat { DescriptorStat { - type_: descriptor_type_for(kind), - link_count: 1, + type_: descriptor_type_for(attrs.kind), + link_count: u64::from(attrs.nlink), size: attrs.size, - // Without a separate atime field in our cache, fall back to mtime. - data_access_timestamp: Some(dt), - data_modification_timestamp: Some(dt), - status_change_timestamp: Some(dt), + data_access_timestamp: Some(systemtime_to_wit(attrs.atime)), + data_modification_timestamp: Some(systemtime_to_wit(attrs.mtime)), + status_change_timestamp: Some(systemtime_to_wit(attrs.ctime)), } } @@ -95,10 +97,17 @@ fn systemtime_to_wit( /// - `NoChange` → `None` (don't update this field) /// - `Now` → `Some(now)` /// - `Timestamp(dt)` → `Some(epoch + dt.seconds + dt.nanoseconds)` -fn wit_timestamp_to_systemtime(ts: NewTimestamp) -> Option { +fn wit_timestamp_to_systemtime( + ts: NewTimestamp, + clock: &crate::clock::WallClockAdapter, +) -> Option { + use wasmtime_wasi::HostWallClock; match ts { NewTimestamp::NoChange => None, - NewTimestamp::Now => Some(std::time::SystemTime::now()), + // The runtime's clock, not the host's: a guest that reads + // `wall-clock` and then calls `set-times` with "now" must not see two + // different answers. + NewTimestamp::Now => Some(std::time::UNIX_EPOCH + clock.now()), NewTimestamp::Timestamp(dt) => { Some(std::time::UNIX_EPOCH + std::time::Duration::new(dt.seconds, dt.nanoseconds)) } @@ -127,14 +136,14 @@ fn split_at_path(p: &str) -> (&str, &str) { async fn resolve_parent( fs: &s3fs_core::Fs, + scope: &std::sync::Arc, base: &std::sync::Arc, parent_path: &str, ) -> S3WasiFsResult> { if parent_path.is_empty() { Ok(base.clone()) } else { - fs.tree - .lookup_at(base, parent_path) + fs.lookup_within(scope, base, parent_path) .await .map_err(|e| S3WasiFsError::from(from_fs(e))) } @@ -152,11 +161,11 @@ impl HostDescriptor for S3FsCtxView<'_> { ) -> Result, S3WasiFsError> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let handle = match d { - Descriptor::File { handle } => handle, + Descriptor::File { handle, .. } => handle, Descriptor::Dir { .. } => return Err(S3WasiFsError::from(ErrorCode::IsDirectory)), }; let s: wasmtime_wasi::p2::DynInputStream = Box::new( - crate::streams::S3InputStream::read_at(self.fs.clone(), handle, offset), + crate::wasi::streams::S3InputStream::read_at(self.fs.clone(), handle, offset), ); let res = self .table @@ -172,11 +181,11 @@ impl HostDescriptor for S3FsCtxView<'_> { ) -> Result, S3WasiFsError> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let handle = match d { - Descriptor::File { handle } => handle, + Descriptor::File { handle, .. } => handle, Descriptor::Dir { .. } => return Err(S3WasiFsError::from(ErrorCode::IsDirectory)), }; let s: wasmtime_wasi::p2::DynOutputStream = Box::new( - crate::streams::S3OutputStream::write_at(self.fs.clone(), handle, offset), + crate::wasi::streams::S3OutputStream::write_at(self.fs.clone(), handle, offset), ); let res = self .table @@ -191,12 +200,12 @@ impl HostDescriptor for S3FsCtxView<'_> { ) -> Result, S3WasiFsError> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let handle = match d { - Descriptor::File { handle } => handle, + Descriptor::File { handle, .. } => handle, Descriptor::Dir { .. } => return Err(S3WasiFsError::from(ErrorCode::IsDirectory)), }; - let offset = *handle.size.read(); + let offset = handle.size().await; let s: wasmtime_wasi::p2::DynOutputStream = Box::new( - crate::streams::S3OutputStream::write_at(self.fs.clone(), handle, offset), + crate::wasi::streams::S3OutputStream::write_at(self.fs.clone(), handle, offset), ); let res = self .table @@ -218,7 +227,7 @@ impl HostDescriptor for S3FsCtxView<'_> { async fn sync_data(&mut self, fd: Resource) -> S3WasiFsResult<()> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; match d { - Descriptor::File { handle } => self.fs.sync(&handle).await.into_wasi(), + Descriptor::File { handle, .. } => self.fs.sync(&handle).await.into_wasi(), Descriptor::Dir { .. } => Err(S3WasiFsError::from(ErrorCode::IsDirectory)), } } @@ -228,12 +237,14 @@ impl HostDescriptor for S3FsCtxView<'_> { } async fn get_type(&mut self, fd: Resource) -> S3WasiFsResult { - let d = self + let inode = self .table .get(&fd) - .map_err(|e| S3WasiFsError::trap(wasmtime::Error::from(e)))?; - let kind = d.inode().kind.read().clone(); - Ok(descriptor_type_for(&kind)) + .map_err(|e| S3WasiFsError::trap(wasmtime::Error::from(e)))? + .inode() + .clone(); + let attrs = self.fs.stat(&inode).await.map_err(from_fs)?; + Ok(descriptor_type_for(attrs.kind)) } async fn set_size( @@ -243,7 +254,7 @@ impl HostDescriptor for S3FsCtxView<'_> { ) -> S3WasiFsResult<()> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let handle = match d { - Descriptor::File { handle } => handle, + Descriptor::File { handle, .. } => handle, Descriptor::Dir { .. } => return Err(S3WasiFsError::from(ErrorCode::IsDirectory)), }; self.fs.set_size(&handle, size).await.into_wasi() @@ -257,11 +268,11 @@ impl HostDescriptor for S3FsCtxView<'_> { ) -> S3WasiFsResult<()> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let handle = match d { - Descriptor::File { handle } => handle, + Descriptor::File { handle, .. } => handle, Descriptor::Dir { .. } => return Err(S3WasiFsError::from(ErrorCode::IsDirectory)), }; - let atime = wit_timestamp_to_systemtime(data_access_timestamp); - let mtime = wit_timestamp_to_systemtime(data_modification_timestamp); + let atime = wit_timestamp_to_systemtime(data_access_timestamp, self.clock); + let mtime = wit_timestamp_to_systemtime(data_modification_timestamp, self.clock); self.fs.set_times(&handle, atime, mtime).await.into_wasi() } @@ -273,7 +284,7 @@ impl HostDescriptor for S3FsCtxView<'_> { ) -> S3WasiFsResult<(Vec, bool)> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let handle = match d { - Descriptor::File { handle } => handle, + Descriptor::File { handle, .. } => handle, Descriptor::Dir { .. } => return Err(S3WasiFsError::from(ErrorCode::IsDirectory)), }; let bytes = self @@ -293,7 +304,7 @@ impl HostDescriptor for S3FsCtxView<'_> { ) -> S3WasiFsResult { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let handle = match d { - Descriptor::File { handle } => handle, + Descriptor::File { handle, .. } => handle, Descriptor::Dir { .. } => return Err(S3WasiFsError::from(ErrorCode::IsDirectory)), }; let n = self @@ -337,18 +348,19 @@ impl HostDescriptor for S3FsCtxView<'_> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let base = d.at_base(); let (parent_path, name) = split_at_path(&path); - let parent = resolve_parent(self.fs, &base, parent_path).await?; + let parent = resolve_parent(self.fs, self.scope, &base, parent_path).await?; self.fs.mkdir(&parent, name).await.map(|_| ()).into_wasi() } async fn stat(&mut self, fd: Resource) -> S3WasiFsResult { - let d = self + let inode = self .table .get(&fd) - .map_err(|e| S3WasiFsError::trap(wasmtime::Error::from(e)))?; - let kind = d.inode().kind.read().clone(); - let attrs = d.inode().attrs.read().clone(); - Ok(stat_from_attrs(&attrs, &kind)) + .map_err(|e| S3WasiFsError::trap(wasmtime::Error::from(e)))? + .inode() + .clone(); + let attrs = self.fs.stat(&inode).await.map_err(from_fs)?; + Ok(stat_from_attrs(&attrs)) } async fn stat_at( @@ -359,15 +371,13 @@ impl HostDescriptor for S3FsCtxView<'_> { ) -> S3WasiFsResult { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let base = d.at_base(); - let target = if path_flags.contains(PathFlags::SYMLINK_FOLLOW) { - self.fs.tree.lookup_at(&base, &path).await - } else { - self.fs.tree.lookup_at_no_follow(&base, &path).await - } - .map_err(|e| S3WasiFsError::from(from_fs(e)))?; - let kind = target.kind.read().clone(); - let attrs = target.attrs.read().clone(); - Ok(stat_from_attrs(&attrs, &kind)) + let follow = path_flags.contains(PathFlags::SYMLINK_FOLLOW); + let attrs = self + .fs + .stat_at(&base, &path, follow) + .await + .map_err(from_fs)?; + Ok(stat_from_attrs(&attrs)) } async fn set_times_at( @@ -381,24 +391,41 @@ impl HostDescriptor for S3FsCtxView<'_> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let base = d.at_base(); let follow = path_flags.contains(PathFlags::SYMLINK_FOLLOW); - let atime = wit_timestamp_to_systemtime(data_access_timestamp); - let mtime = wit_timestamp_to_systemtime(data_modification_timestamp); + let atime = wit_timestamp_to_systemtime(data_access_timestamp, self.clock); + let mtime = wit_timestamp_to_systemtime(data_modification_timestamp, self.clock); self.fs - .set_times_at(&base, &path, follow, atime, mtime) + .set_times_at(&base, &path, atime, mtime, follow) .await .into_wasi() } async fn link_at( &mut self, - _fd: Resource, - _old_path_flags: PathFlags, - _old_path: String, - _new_descriptor: Resource, - _new_path: String, + fd: Resource, + old_path_flags: PathFlags, + old_path: String, + new_descriptor: Resource, + new_path: String, ) -> S3WasiFsResult<()> { - // Hardlinks: documented as ❌ in the Compatibility Matrix. - Err(S3WasiFsError::from(ErrorCode::Unsupported)) + let old_d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; + let new_d = + get_descriptor_owned(self.table, &new_descriptor).map_err(S3WasiFsError::trap)?; + let old_base = old_d.at_base(); + let new_base = new_d.at_base(); + let (parent_path, name) = split_at_path(&new_path); + let new_parent = resolve_parent(self.fs, self.scope, &new_base, parent_path).await?; + + self.fs + .link_within( + self.scope, + &old_base, + &old_path, + &new_parent, + name, + old_path_flags.contains(PathFlags::SYMLINK_FOLLOW), + ) + .await + .into_wasi() } async fn open_at( @@ -423,21 +450,31 @@ impl HostDescriptor for S3FsCtxView<'_> { if !path_flags.contains(PathFlags::SYMLINK_FOLLOW) { // Probe the leaf without following. NotFound is fine — open_at // may be creating. Actual symlink → Loop. - match self.fs.tree.lookup_at_no_follow(&base, &path).await { - Ok(probe) if probe.is_symlink() => { + if let Ok(probe) = self + .fs + .lookup_within_no_follow(self.scope, &base, &path) + .await + { + if matches!( + self.fs.stat(&probe).await.map(|a| a.kind), + Ok(InodeKind::Symlink) + ) { return Err(S3WasiFsError::from(ErrorCode::Loop)); } - _ => {} } } let handle = self .fs - .open_at(&base, &path, flags) + .open_within(self.scope, &base, &path, flags) .await .map_err(|e| S3WasiFsError::from(from_fs(e)))?; - let descriptor = if handle.inode.is_dir() { + let is_dir = matches!( + self.fs.stat(&handle.inode).await.map_err(from_fs)?.kind, + InodeKind::Directory + ); + let descriptor = if is_dir { // Don't keep a file handle for a directory. self.fs .close(&handle) @@ -454,7 +491,14 @@ impl HostDescriptor for S3FsCtxView<'_> { .map_err(|e| S3WasiFsError::from(from_fs(e)))?; return Err(S3WasiFsError::from(ErrorCode::NotDirectory)); } - Descriptor::File { handle } + Descriptor::File { + parent: self + .fs + .parent_within(self.scope, &handle.inode) + .await + .map_err(from_fs)?, + handle, + } }; let res = self @@ -472,7 +516,7 @@ impl HostDescriptor for S3FsCtxView<'_> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let base = d.at_base(); let (parent_path, name) = split_at_path(&path); - let parent = resolve_parent(self.fs, &base, parent_path).await?; + let parent = resolve_parent(self.fs, self.scope, &base, parent_path).await?; self.fs.readlink_at(&parent, name).await.into_wasi() } @@ -484,7 +528,7 @@ impl HostDescriptor for S3FsCtxView<'_> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let base = d.at_base(); let (parent_path, name) = split_at_path(&path); - let parent = resolve_parent(self.fs, &base, parent_path).await?; + let parent = resolve_parent(self.fs, self.scope, &base, parent_path).await?; self.fs.rmdir(&parent, name).await.into_wasi() } @@ -502,8 +546,8 @@ impl HostDescriptor for S3FsCtxView<'_> { let new_base = new_d.at_base(); let (op, oname) = split_at_path(&old_path); let (np, nname) = split_at_path(&new_path); - let old_parent = resolve_parent(self.fs, &old_base, op).await?; - let new_parent = resolve_parent(self.fs, &new_base, np).await?; + let old_parent = resolve_parent(self.fs, self.scope, &old_base, op).await?; + let new_parent = resolve_parent(self.fs, self.scope, &new_base, np).await?; self.fs .rename(&old_parent, oname, &new_parent, nname) .await @@ -519,7 +563,7 @@ impl HostDescriptor for S3FsCtxView<'_> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let base = d.at_base(); let (parent_path, name) = split_at_path(&new_path); - let parent = resolve_parent(self.fs, &base, parent_path).await?; + let parent = resolve_parent(self.fs, self.scope, &base, parent_path).await?; self.fs .symlink_at(&parent, name, &old_path) .await @@ -535,7 +579,7 @@ impl HostDescriptor for S3FsCtxView<'_> { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let base = d.at_base(); let (parent_path, name) = split_at_path(&path); - let parent = resolve_parent(self.fs, &base, parent_path).await?; + let parent = resolve_parent(self.fs, self.scope, &base, parent_path).await?; self.fs.unlink(&parent, name).await.into_wasi() } @@ -544,8 +588,10 @@ impl HostDescriptor for S3FsCtxView<'_> { a: Resource, b: Resource, ) -> Result { - let da = self.table.get(&a)?.inode().id; - let db = self.table.get(&b)?.inode().id; + // Object ids are never reused and the generation distinguishes a + // reused slot, so this is exact rather than a hash of an ETag. + let da = **self.table.get(&a)?.inode(); + let db = **self.table.get(&b)?.inode(); Ok(da == db) } @@ -553,11 +599,13 @@ impl HostDescriptor for S3FsCtxView<'_> { &mut self, fd: Resource, ) -> S3WasiFsResult { - let d = self + let inode = self .table .get(&fd) - .map_err(|e| S3WasiFsError::trap(wasmtime::Error::from(e)))?; - Ok(metadata_hash_for(d)) + .map_err(|e| S3WasiFsError::trap(wasmtime::Error::from(e)))? + .inode() + .clone(); + Ok(metadata_hash_for(&inode)) } async fn metadata_hash_at( @@ -568,45 +616,35 @@ impl HostDescriptor for S3FsCtxView<'_> { ) -> S3WasiFsResult { let d = get_descriptor_owned(self.table, &fd).map_err(S3WasiFsError::trap)?; let base = d.at_base(); + let follow = _path_flags.contains(PathFlags::SYMLINK_FOLLOW); let target = self .fs - .tree - .lookup_at(&base, &path) + .resolve_for_hash(&base, &path, follow) .await - .map_err(|e| S3WasiFsError::from(from_fs(e)))?; - let attrs = target.attrs.read(); - Ok(MetadataHashValue { - lower: simple_hash(&attrs.etag, attrs.size), - upper: attrs.size, - }) + .map_err(from_fs)?; + Ok(metadata_hash_for(&target)) } async fn drop(&mut self, fd: Resource) -> Result<()> { let d = self.table.delete(fd)?; - if let Descriptor::File { handle } = d { - // Best-effort close. Any in-flight MPU is aborted. + if let Descriptor::File { handle, .. } = d { + // Best-effort close, which flushes a writable handle. let _ = self.fs.close(&handle).await; } Ok(()) } } -fn metadata_hash_for(d: &Descriptor) -> MetadataHashValue { - let attrs = d.inode().attrs.read(); +/// Identity, not content: the object id and its generation. WASI only +/// requires that the value differ when the objects differ, and this pair is +/// exactly that — stable across mounts, and never recycled. +fn metadata_hash_for(inode: &Inode) -> MetadataHashValue { MetadataHashValue { - lower: simple_hash(&attrs.etag, attrs.size), - upper: attrs.size, + lower: inode.objid(), + upper: inode.gen, } } -fn simple_hash(etag: &str, size: u64) -> u64 { - use std::hash::{Hash, Hasher}; - let mut h = std::collections::hash_map::DefaultHasher::new(); - etag.hash(&mut h); - size.hash(&mut h); - h.finish() -} - // --------------------------------------------------------------------------- // HostDirectoryEntryStream // --------------------------------------------------------------------------- @@ -624,9 +662,8 @@ impl HostDirectoryEntryStream for S3FsCtxView<'_> { return Ok(None); } let entry = &s.entries[s.cursor]; - let kind = entry.inode.kind.read().clone(); let dir_entry = DirectoryEntry { - type_: descriptor_type_for(&kind), + type_: descriptor_type_for(entry.kind), name: entry.name.clone(), }; s.cursor += 1; diff --git a/runtime/src/wasi/host_preopens.rs b/runtime/src/wasi/host_preopens.rs new file mode 100644 index 0000000..047c12b --- /dev/null +++ b/runtime/src/wasi/host_preopens.rs @@ -0,0 +1,21 @@ +//! `wasi:filesystem/preopens::Host` — exposes the single root descriptor. + +use wasmtime::component::Resource; +use wasmtime::Result; + +use crate::wasi::bindings::wasi::filesystem::preopens::Host; +use crate::wasi::bindings::wasi::filesystem::types::Descriptor as WitDescriptor; +use crate::wasi::descriptors::Descriptor; +use crate::wasi::view::S3FsCtxView; + +impl Host for S3FsCtxView<'_> { + async fn get_directories(&mut self) -> Result, String)>> { + // The scope, not the filesystem root. This is the only descriptor a + // guest is ever handed for free, so it is the only place a tenant's + // view of `/` has to be established — everything else is derived from + // it by resolution that cannot leave it. + let root = self.scope.clone(); + let descriptor = self.table.push(Descriptor::Dir { inode: root })?; + Ok(vec![(descriptor, self.fs.config.mount_path.clone())]) + } +} diff --git a/runtime/src/wasi/mod.rs b/runtime/src/wasi/mod.rs new file mode 100644 index 0000000..1800145 --- /dev/null +++ b/runtime/src/wasi/mod.rs @@ -0,0 +1,46 @@ +//! `wasi:filesystem@0.2.x` implemented over `s3fs-core`'s `Fs`. +//! +//! Only `wasi:filesystem/{types,preopens}` lives here. Everything else a guest +//! needs — `wasi:io`, `wasi:cli`, clocks, random, sockets — comes from +//! `wasmtime-wasi` unchanged, and [`crate::linker`] assembles the two. +//! +//! This module depends on nothing AWS-specific: it is written against +//! `s3fs_core::Fs`, so it works over any [`s3fs_core::backend::Backend`], +//! including the in-memory one. That is why `mount` — the only part that needs +//! the AWS SDK — sits behind a feature flag rather than here. + +pub mod bindings; +pub mod descriptors; +pub mod error_map; +pub mod host_filesystem; +pub mod host_preopens; +pub mod streams; +pub mod view; + +pub use descriptors::{Descriptor, DirectoryEntryStream}; +pub use view::{S3FsCtxView, S3WasiView}; + +use wasmtime::component::{HasData, Linker}; +use wasmtime::Result; + +/// `HasData` marker so bindgen knows the trait impls live on +/// [`S3FsCtxView<'_>`]. +pub struct HasS3Fs; +impl HasData for HasS3Fs { + type Data<'a> = S3FsCtxView<'a>; +} + +/// Add `wasi:filesystem/{types,preopens}` to `linker`, backed by the `Fs` the +/// store yields through [`S3WasiView`]. +/// +/// Public and generic over `T` so a host with its own state type can take just +/// the filesystem and build the rest of its linker however it likes. +/// [`crate::build_linker`] is the batteries-included version. +pub fn add_filesystem_to_linker(linker: &mut Linker) -> Result<()> { + fn getter(t: &mut T) -> S3FsCtxView<'_> { + t.s3fs_view() + } + bindings::wasi::filesystem::types::add_to_linker::(linker, getter::)?; + bindings::wasi::filesystem::preopens::add_to_linker::(linker, getter::)?; + Ok(()) +} diff --git a/crates/s3fs-wasmtime/src/streams.rs b/runtime/src/wasi/streams.rs similarity index 99% rename from crates/s3fs-wasmtime/src/streams.rs rename to runtime/src/wasi/streams.rs index ae95798..9510610 100644 --- a/crates/s3fs-wasmtime/src/streams.rs +++ b/runtime/src/wasi/streams.rs @@ -13,7 +13,7 @@ use bytes::Bytes; use s3fs_core::{FileHandle, Fs}; use wasmtime_wasi::p2::{InputStream, OutputStream, Pollable, StreamError, StreamResult}; -use crate::error_map::from_fs; +use crate::wasi::error_map::from_fs; /// Soft cap on `check_write` permits. Big enough to fit a typical write /// buffer in one shot. diff --git a/crates/s3fs-wasmtime/src/view.rs b/runtime/src/wasi/view.rs similarity index 57% rename from crates/s3fs-wasmtime/src/view.rs rename to runtime/src/wasi/view.rs index 5cf6cd7..2a15239 100644 --- a/crates/s3fs-wasmtime/src/view.rs +++ b/runtime/src/wasi/view.rs @@ -3,7 +3,7 @@ use std::sync::Arc; -use s3fs_core::Fs; +use s3fs_core::{Fs, Inode}; use wasmtime::component::ResourceTable; /// "View" struct holding mutable references the host trait impls need. @@ -18,10 +18,21 @@ use wasmtime::component::ResourceTable; /// writes them. pub struct S3FsCtxView<'a> { pub fs: &'a Arc, + /// What this guest sees as `/`. + /// + /// The filesystem root for a guest that has the whole store; a tenant's + /// own directory when clients are separated inside one filesystem. Every + /// path this view resolves is confined to it — see + /// [`s3fs_core::Fs::lookup_within`] — so separation is a property of the + /// capability layer rather than something guest code is trusted to keep. + pub scope: &'a Arc, pub table: &'a mut ResourceTable, + /// The same clock the guest sees through `wasi:clocks/wall-clock`, so + /// `set-times` with "now" cannot disagree with what the guest just read. + pub clock: &'a Arc, } -/// Implement this on your store-data type `T` to plug `s3fs-wasmtime` into +/// Implement this on your store-data type `T` to plug the filesystem into /// a `Linker`. The closure passed to [`crate::add_to_linker`] uses this /// trait to fetch a fresh view per call. pub trait S3WasiView: Send { diff --git a/runtime/tests/boot_origin.rs b/runtime/tests/boot_origin.rs new file mode 100644 index 0000000..26a9515 --- /dev/null +++ b/runtime/tests/boot_origin.rs @@ -0,0 +1,856 @@ +//! The boot machine: which state an enclave will load, and what it records. +//! +//! `Store::open` used to create a filesystem when it found none, so an enclave +//! pointed at an emptied store served a fresh, correctly-signed, completely +//! wrong filesystem. Most of these tests are about the *refusals* that replaced +//! that, because the happy path was never the problem. The rest are about pair +//! records: what a new runtime or guest leaves behind, and what a restart does +//! not. +//! +//! Everything runs over `MemoryBackend`, so there is no Docker and no network: +//! what is under test is the decision, not S3. +//! +//! **Receipts here are unsigned**, and deliberately. `ReceiptTrust::Required` +//! would demand a chain to the real AWS root, which no test can produce. +//! `UnsignedEmulator` skips the *signature* and still applies every +//! expectation — PCR0, PCR16, `user_data` — so the state machine, the identity +//! binding and the pair records are all genuinely exercised. The signature path +//! is `nitro-attestation`'s own tests, against real ES384 chains. +//! +//! The NSM here keeps registers the way the device does: 0–15 locked from the +//! start, 16 free until the guest is measured, and **only locked registers in a +//! document**. The fake it replaced put an unlocked register into every +//! document, which is how a successor handoff that could never have worked on +//! hardware passed here. + +use std::ops::Range; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::{Arc, Mutex}; + +use bytes::Bytes; +use enclave_runtime::boot::{BootConfig, BootMode, Pair, ReceiptTrust}; +use enclave_runtime::{Backends, MasterKeySource, MountConfig, SealedKey, StaticKey}; +use nitro_attestation::testing::TestChain; +use nitro_attestation::AttestationDocument; +use nitro_nsm::{AttestationRequest, Nsm, Pcr, PCR_GUEST, PCR_ZERO}; +use s3fs_core::backend::{ + memory::MemoryBackend, Backend, BlobMeta, Capabilities, CompletedPart, CopyBlobInput, + GetBlobOutput, ListBlobsInput, ListBlobsOutput, MultipartId, PartUploadOutput, PutBlobInput, +}; +use s3fs_core::{FsError, MasterSecret}; + +const GUEST_A: &[u8] = b"guest component A"; +const GUEST_B: &[u8] = b"guest component B"; +const FS_ID: [u8; 16] = [3u8; 16]; + +/// An NSM that keeps registers like the device, and signs documents listing +/// the locked ones. +struct TestNsm { + chain: TestChain, + pcrs: Mutex>, + attestations: AtomicUsize, +} + +impl std::fmt::Debug for TestNsm { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str("TestNsm") + } +} + +impl TestNsm { + /// A fresh enclave running image `pcr0`, before its runtime has measured + /// anything. + fn unmeasured(pcr0: u8) -> Arc { + Arc::new(TestNsm { + chain: TestChain::new().expect("chain"), + pcrs: Mutex::new( + (0..32) + .map(|i| Pcr { + locked: i < 16, + value: if i == 0 { + vec![pcr0; 48] + } else { + PCR_ZERO.to_vec() + }, + }) + .collect(), + ), + attestations: AtomicUsize::new(0), + }) + } + + /// A fresh enclave running image `pcr0`, with `guest` measured the way the + /// runtime measures it. A restart is one of these, not the same one again. + fn running(pcr0: u8, guest: &[u8]) -> Arc { + let nsm = Self::unmeasured(pcr0); + enclave_runtime::measure_guest(nsm.as_ref(), guest).expect("measuring the guest"); + nsm + } + + fn attestations(&self) -> usize { + self.attestations.load(Ordering::SeqCst) + } +} + +impl Nsm for TestNsm { + fn get_random(&self, buf: &mut [u8]) -> anyhow::Result<()> { + buf.fill(0x5a); + Ok(()) + } + + fn attest(&self, request: &AttestationRequest) -> anyhow::Result> { + self.attestations.fetch_add(1, Ordering::SeqCst); + let pcrs = self + .pcrs + .lock() + .unwrap() + .iter() + .enumerate() + .filter(|(_, pcr)| pcr.locked) + .map(|(index, pcr)| (index as u32, pcr.value.clone())) + .collect(); + self.chain + .document_with_pcrs(request.user_data.clone(), request.nonce.clone(), pcrs) + } + + fn describe_pcr(&self, index: u16) -> anyhow::Result { + self.pcrs + .lock() + .unwrap() + .get(index as usize) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no PCR{index}")) + } + + fn extend_pcr(&self, index: u16, data: &[u8]) -> anyhow::Result> { + let mut pcrs = self.pcrs.lock().unwrap(); + let pcr = pcrs + .get_mut(index as usize) + .ok_or_else(|| anyhow::anyhow!("no PCR{index}"))?; + anyhow::ensure!(!pcr.locked, "PCR{index} is read-only"); + pcr.value = nitro_nsm::pcr_extend(&pcr.value, data); + Ok(pcr.value.clone()) + } + + fn lock_pcr(&self, index: u16) -> anyhow::Result<()> { + let mut pcrs = self.pcrs.lock().unwrap(); + let pcr = pcrs + .get_mut(index as usize) + .ok_or_else(|| anyhow::anyhow!("no PCR{index}"))?; + anyhow::ensure!(!pcr.locked, "PCR{index} is read-only"); + pcr.locked = true; + Ok(()) + } + + fn describe(&self) -> String { + "test NSM".into() + } +} + +/// One store, reused across boots, so "resume" means what it says. +struct Store { + data: Arc, + roots: Arc, +} + +impl Store { + fn new() -> Self { + Store { + data: Arc::new(MemoryBackend::new()), + roots: Arc::new(MemoryBackend::new()), + } + } + + fn backends(&self) -> Backends { + Backends { + data: self.data.clone(), + roots: self.roots.clone(), + } + } + + /// Write an object with no retention, to construct a store in a shape a + /// completed genesis never produces. + async fn put(&self, key: &str, body: Vec) { + self.roots + .put_blob(PutBlobInput::new(key, body.into())) + .await + .expect("write"); + } + + /// Every origin record, sorted. What a boot wrote is the difference + /// between two of these. + async fn origin_records(&self) -> Vec { + let listed = self + .roots + .list_blobs(ListBlobsInput { + prefix: "origin/", + max_keys: Some(1000), + start_after: None, + continuation_token: None, + delimiter: None, + }) + .await + .expect("list"); + let mut keys: Vec = listed.items.into_iter().map(|item| item.key).collect(); + keys.sort(); + keys + } + + async fn pair_records(&self) -> Vec { + self.origin_records() + .await + .into_iter() + .filter(|key| key.contains(".pair.")) + .collect() + } + + async fn document(&self, key: &str) -> AttestationDocument { + let body = self + .roots + .get_retained_blob(key) + .await + .expect("the record") + .body; + nitro_attestation::parse(&body).expect("a document") + } +} + +fn config() -> MountConfig { + MountConfig { + bucket: "data".into(), + roots_bucket: Some("roots".into()), + region: "us-east-1".into(), + endpoint: None, + access_key_id: None, + secret_access_key: None, + session_token: None, + force_path_style: false, + bucket_prefix: String::new(), + mount_path: "/".into(), + fs_id: FS_ID, + min_root_seq: None, + skip_bucket_probe: true, + request_timeout: std::time::Duration::from_secs(30), + } +} + +fn boot_config() -> BootConfig { + BootConfig { + trust: ReceiptTrust::UnsignedEmulator, + } +} + +fn key() -> StaticKey { + StaticKey::new(MasterSecret::from_bytes([8u8; 32])) +} + +fn pair_of(pcr0: u8, guest: &[u8]) -> Pair { + Pair { + pcr0: [pcr0; 48], + pcr16: nitro_attestation::guest_pcr(guest), + } +} + +fn record_key(pair: &Pair) -> String { + pair.record_key("", &FS_ID) +} + +async fn boot(store: &Store, nsm: &Arc) -> anyhow::Result { + boot_on(store.backends(), nsm).await +} + +async fn boot_on( + backends: Backends, + nsm: &Arc, +) -> anyhow::Result { + let nsm: Arc = nsm.clone(); + enclave_runtime::boot(&backends, &config(), &boot_config(), &nsm, &key()).await +} + +#[tokio::test] +async fn an_empty_store_becomes_a_filesystem_with_an_attested_origin() { + let store = Store::new(); + let booted = boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + assert_eq!(booted.mode, BootMode::Genesis); + assert_eq!(booted.pair, pair_of(0xaa, GUEST_A)); + + // The receipt, the sealed key, and the record of who ran genesis. + let pair_key = record_key(&booted.pair); + let mut expected = vec![ + format!("origin/{}.key", hex::encode(FS_ID)), + format!("origin/{}.receipt", hex::encode(FS_ID)), + pair_key.clone(), + ]; + expected.sort(); + assert_eq!(store.origin_records().await, expected); + + // It says who, in registers a document carries only once they are locked. + let record = store.document(&pair_key).await; + assert_eq!(record.pcr(0), Some(&[0xaa; 48][..])); + assert_eq!( + record.pcr(PCR_GUEST as u32), + Some(&nitro_attestation::guest_pcr(GUEST_A)[..]) + ); + + // And it stays said. + assert!( + store + .roots + .put_blob(PutBlobInput::new(&pair_key, b"rewritten".to_vec().into())) + .await + .is_err(), + "COMPLIANCE retention must refuse to overwrite a pair record" + ); +} + +/// The property the whole milestone is for: a second boot recognises the state +/// as its own, rather than making a new one. And a plain restart leaves no +/// trace — records accumulate per pair, not per boot. +#[tokio::test] +async fn a_restart_resumes_the_same_state_and_writes_nothing() { + let store = Store::new(); + let first = boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + let before = store.origin_records().await; + + let restarted = TestNsm::running(0xaa, GUEST_A); + let second = boot(&store, &restarted).await.expect("resume"); + + assert_eq!(second.mode, BootMode::Resume); + assert_eq!( + second.state_root, first.state_root, + "the same filesystem must hash to the same state_root" + ); + assert_eq!( + store.origin_records().await, + before, + "a restart wrote an origin record" + ); + assert_eq!( + restarted.attestations(), + 0, + "a restart asked the NSM for a record it already had" + ); +} + +/// **The delete marker.** Object Lock protects a *version*; it does not stop a +/// `DeleteObject` without a version id, which writes a marker that hides the +/// object from every ordinary read while the bytes stay undeletable underneath. +/// +/// That gap is only dangerous where absence means something. Hide the receipt +/// *and* the sealed key and the naive reading is `(None, None)` — create a +/// filesystem — while the real one sits in the same bucket. Hide the pair +/// record and the naive reading is "first boot of this pair". The enclave reads +/// the retained versions instead, so the records are still found, and the boot +/// is still a plain resume. +#[tokio::test] +async fn hiding_the_origin_records_behind_a_delete_marker_does_not_work() { + let store = Store::new(); + let first = boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + + let hidden = [ + format!("origin/{}.receipt", hex::encode(FS_ID)), + format!("origin/{}.key", hex::encode(FS_ID)), + record_key(&first.pair), + ]; + for key in &hidden { + // Succeeds, as S3 does. Refusing would be the comfortable answer and + // is not the true one. + store + .roots + .delete_blob(key) + .await + .expect("S3 allows a delete marker"); + assert!( + store.roots.get_blob(key, None).await.is_err(), + "{key} should be gone as far as an ordinary read can tell" + ); + } + + let second = boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("the records are still there"); + assert_eq!(second.mode, BootMode::Resume); + assert_eq!( + second.state_root, first.state_root, + "a hidden origin must not become a new one" + ); +} + +/// The same attack against a store that reads the *current* version, to show +/// the test above is not passing for an unrelated reason. This is what the +/// enclave used to do, and it ends in a second filesystem. +#[tokio::test] +async fn reading_the_current_version_is_what_made_hiding_work() { + let store = Store::new(); + boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + + let receipt = format!("origin/{}.receipt", hex::encode(FS_ID)); + store.roots.delete_blob(&receipt).await.unwrap(); + + // What `get_blob` — the old read — would have reported. + assert!( + store.roots.get_blob(&receipt, None).await.is_err(), + "the marker hides it" + ); + // What the conditional PUT would then have allowed: the genesis lease is + // free again, so a second enclave could take it. + let retaken = store + .roots + .put_blob_if_not_exists(PutBlobInput::new( + receipt.clone(), + Bytes::from_static(b"a forged origin"), + )) + .await; + assert!( + retaken.is_ok(), + "the lease is retakeable over a marker — this is why absence must be \ + read from the retained version, not the current one" + ); + + // And the retained version is still the real one underneath. + let retained = store.roots.get_retained_blob(&receipt).await.unwrap(); + assert_ne!( + retained.body.as_ref(), + b"a forged origin", + "Object Lock kept the original version" + ); +} + +/// **The state substitution.** A receipt with no filesystem under it: either +/// the store is hiding everything, or it is not the store the receipt +/// describes. Either way there is nothing here this enclave may serve, and the +/// old code would have made a fresh empty filesystem instead. +/// +/// Written directly rather than by hiding, because the test above shows hiding +/// no longer produces this state — this is the shape, not the route to it. +#[tokio::test] +async fn a_receipt_without_a_filesystem_is_refused() { + let store = Store::new(); + let nsm = TestNsm::running(0xaa, GUEST_A); + + store + .put( + &format!("origin/{}.receipt", hex::encode(FS_ID)), + nsm.attest(&AttestationRequest::with_user_data(b"anything".to_vec())) + .unwrap(), + ) + .await; + + let err = boot(&store, &nsm).await.expect_err("must refuse"); + let text = format!("{err:#}"); + assert!(text.contains("no sealed key"), "{text}"); + assert!( + text.contains("Refusing to boot"), + "the refusal must be explicit about what it is not doing: {text}" + ); +} + +/// The mirror: a sealed key with no receipt. That is what an interrupted +/// genesis leaves — the receipt is written last, on purpose — and it is also +/// what hiding the receipt leaves. Nothing on this side can tell those apart, +/// so the only safe answer is to refuse both. +#[tokio::test] +async fn a_sealed_key_without_a_receipt_is_refused() { + let store = Store::new(); + let (_, sealed) = key().mint().await.unwrap(); + store + .put( + &format!("origin/{}.key", hex::encode(FS_ID)), + sealed.as_bytes().to_vec(), + ) + .await; + + let err = boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect_err("must refuse"); + let text = format!("{err:#}"); + assert!(text.contains("no state-origin receipt"), "{text}"); + assert!(text.contains("Refusing to boot"), "{text}"); +} + +/// A key source that counts what it was asked for. +#[derive(Debug)] +struct CountingKey { + inner: StaticKey, + asked: AtomicUsize, +} + +#[async_trait::async_trait] +impl MasterKeySource for CountingKey { + fn describe(&self) -> &'static str { + "counting test key" + } + + async fn mint(&self) -> anyhow::Result<(MasterSecret, SealedKey)> { + self.asked.fetch_add(1, Ordering::SeqCst); + self.inner.mint().await + } + + async fn open(&self, sealed: &SealedKey) -> anyhow::Result { + self.asked.fetch_add(1, Ordering::SeqCst); + self.inner.open(sealed).await + } +} + +/// **The ordering.** A guest that was never measured must not get as far as +/// KMS: the attestation it presented would carry no guest, and nothing would +/// stop the register being extended after a key was released. +#[tokio::test] +async fn an_unmeasured_guest_is_refused_before_any_key_is_asked_for() { + let store = Store::new(); + let nsm: Arc = TestNsm::unmeasured(0xaa); + let keys = CountingKey { + inner: key(), + asked: AtomicUsize::new(0), + }; + + let err = enclave_runtime::boot(&store.backends(), &config(), &boot_config(), &nsm, &keys) + .await + .expect_err("must refuse"); + assert!( + format!("{err:#}").contains("PCR16 is not locked"), + "{err:#}" + ); + assert_eq!( + keys.asked.load(Ordering::SeqCst), + 0, + "a key was asked for on behalf of an unmeasured guest" + ); + assert!( + store.origin_records().await.is_empty(), + "nothing may be written for it either" + ); +} + +/// **A new guest.** The key policy was edited to name it, so the boot machine +/// does not refuse it. What it does is leave a record — exactly one — carrying +/// the new guest's measurement against this state, and nothing on the restarts +/// after. +#[tokio::test] +async fn a_new_guest_is_an_upgrade_and_is_recorded_once() { + let store = Store::new(); + let genesis = boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + + let upgraded = TestNsm::running(0xaa, GUEST_B); + let first = boot(&store, &upgraded).await.expect("upgrade"); + assert_eq!(first.mode, BootMode::Upgrade); + assert_eq!( + first.state_root, genesis.state_root, + "an upgrade must not change what the state is" + ); + assert_eq!( + upgraded.attestations(), + 1, + "one document, for the one record" + ); + + let records = store.pair_records().await; + assert_eq!(records.len(), 2, "{records:?}"); + let record = store.document(&record_key(&pair_of(0xaa, GUEST_B))).await; + assert_eq!(record.pcr(0), Some(&[0xaa; 48][..])); + assert_eq!( + record.pcr(PCR_GUEST as u32), + Some(&nitro_attestation::guest_pcr(GUEST_B)[..]) + ); + + let before = store.origin_records().await; + let again = boot(&store, &TestNsm::running(0xaa, GUEST_B)) + .await + .expect("resume"); + assert_eq!( + again.mode, + BootMode::Resume, + "a restart after an upgrade is not another upgrade" + ); + assert_eq!(store.origin_records().await, before); +} + +/// **A new runtime image.** This used to be refused unless the outgoing image +/// had authorised the incoming one. The key policy decides that now, and the +/// boot machine records it exactly as it records a new guest. +#[tokio::test] +async fn a_new_runtime_image_is_an_upgrade_too() { + let store = Store::new(); + let genesis = boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + + let upgraded = boot(&store, &TestNsm::running(0xbb, GUEST_A)) + .await + .expect("upgrade"); + assert_eq!(upgraded.mode, BootMode::Upgrade); + assert_eq!(upgraded.state_root, genesis.state_root); + assert_eq!(store.pair_records().await.len(), 2); +} + +/// **Set, not sequence.** Returning to a pair that has held the state before +/// finds its record and writes nothing, so the store cannot show that a +/// rollback happened — only that both pairs have held the state. The order is +/// in CloudTrail's record of key-policy edits. Pinned here so the trade-off is +/// visible where it is made. +#[tokio::test] +async fn returning_to_an_earlier_pair_adds_no_record() { + let store = Store::new(); + boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + boot(&store, &TestNsm::running(0xaa, GUEST_B)) + .await + .expect("upgrade"); + + let back = boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("back again"); + assert_eq!(back.mode, BootMode::Resume); + assert_eq!(store.pair_records().await.len(), 2); +} + +/// **A copied record.** A genuine record from another pair, copied to this +/// pair's key ahead of its first boot. It is signed; it is simply about someone +/// else. Found rather than written, and still refused, because a record is +/// checked for what it says and not only for being there. +#[tokio::test] +async fn a_record_copied_from_another_pair_is_refused() { + let store = Store::new(); + boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + + let genuine = store + .roots + .get_retained_blob(&record_key(&pair_of(0xaa, GUEST_A))) + .await + .unwrap() + .body; + store + .put(&record_key(&pair_of(0xaa, GUEST_B)), genuine.to_vec()) + .await; + + let err = boot(&store, &TestNsm::running(0xaa, GUEST_B)) + .await + .expect_err("must refuse"); + let text = format!("{err:#}"); + assert!(text.contains("PCR16 mismatch"), "{text}"); +} + +/// Plants an object at a pair record's key a moment before the enclave's own +/// conditional write lands — the shape of another first boot of the same pair +/// winning the race, or of something less friendly doing the same. +#[derive(Debug)] +struct Interloper { + inner: Arc, + plant: Mutex>, +} + +#[derive(Debug)] +enum Planted { + /// A copy of the record the enclave is about to write: what a concurrent + /// boot of the same pair would have written. + TheSameRecord, + /// These bytes instead. + Bytes(Vec), +} + +impl Interloper { + fn planting(inner: Arc, planted: Planted) -> Arc { + Arc::new(Interloper { + inner, + plant: Mutex::new(Some(planted)), + }) + } +} + +#[async_trait::async_trait] +impl Backend for Interloper { + fn capabilities(&self) -> Capabilities { + self.inner.capabilities() + } + + async fn head_blob(&self, key: &str) -> Result { + self.inner.head_blob(key).await + } + + async fn get_blob( + &self, + key: &str, + range: Option>, + ) -> Result { + self.inner.get_blob(key, range).await + } + + async fn get_retained_blob(&self, key: &str) -> Result { + self.inner.get_retained_blob(key).await + } + + async fn put_blob(&self, input: PutBlobInput) -> Result { + self.inner.put_blob(input).await + } + + async fn put_blob_if_not_exists(&self, input: PutBlobInput) -> Result { + if input.key.contains(".pair.") { + let planted = self.plant.lock().unwrap().take(); + if let Some(planted) = planted { + let body = match planted { + Planted::TheSameRecord => input.body.clone(), + Planted::Bytes(bytes) => bytes.into(), + }; + self.inner + .put_blob_if_not_exists(PutBlobInput::new(input.key.clone(), body)) + .await?; + } + } + self.inner.put_blob_if_not_exists(input).await + } + + async fn delete_blob(&self, key: &str) -> Result<(), FsError> { + self.inner.delete_blob(key).await + } + + async fn list_blobs(&self, input: ListBlobsInput<'_>) -> Result { + self.inner.list_blobs(input).await + } + + async fn copy_blob(&self, input: CopyBlobInput) -> Result { + self.inner.copy_blob(input).await + } + + async fn multipart_begin(&self, input: PutBlobInput) -> Result { + self.inner.multipart_begin(input).await + } + + async fn multipart_upload_part( + &self, + key: &str, + upload_id: &MultipartId, + part_number: u32, + body: Bytes, + ) -> Result { + self.inner + .multipart_upload_part(key, upload_id, part_number, body) + .await + } + + async fn multipart_upload_part_copy( + &self, + key: &str, + upload_id: &MultipartId, + part_number: u32, + source_key: &str, + source_range: Range, + ) -> Result { + self.inner + .multipart_upload_part_copy(key, upload_id, part_number, source_key, source_range) + .await + } + + async fn multipart_complete( + &self, + key: &str, + upload_id: &MultipartId, + parts: Vec, + ) -> Result { + self.inner.multipart_complete(key, upload_id, parts).await + } + + async fn multipart_abort(&self, key: &str, upload_id: &MultipartId) -> Result<(), FsError> { + self.inner.multipart_abort(key, upload_id).await + } +} + +/// **The race.** Two first boots of one pair both find no record, both attest, +/// and one conditional write wins. The loser boots on the winner's record — it +/// says the same thing — rather than failing, and there is still one record. +#[tokio::test] +async fn a_first_boot_that_loses_the_race_boots_on_the_winners_record() { + let store = Store::new(); + boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + + let backends = Backends { + data: store.data.clone(), + roots: Interloper::planting(store.roots.clone(), Planted::TheSameRecord), + }; + let booted = boot_on(backends, &TestNsm::running(0xaa, GUEST_B)) + .await + .expect("the loser boots"); + assert_eq!( + booted.mode, + BootMode::Upgrade, + "it is still this pair's first boot" + ); + assert_eq!(store.pair_records().await.len(), 2, "one record per pair"); +} + +/// **The pre-emption.** Junk lands at the key between the check and the write. +/// The enclave's own record can no longer be stored there, so the junk would +/// stand as the pair's only record — and it is refused rather than accepted as +/// one. +#[tokio::test] +async fn junk_that_wins_the_race_is_refused() { + let store = Store::new(); + boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + + let backends = Backends { + data: store.data.clone(), + roots: Interloper::planting( + store.roots.clone(), + Planted::Bytes(b"not a record".to_vec()), + ), + }; + let err = boot_on(backends, &TestNsm::running(0xaa, GUEST_B)) + .await + .expect_err("must refuse"); + assert!(format!("{err:#}").contains("does not describe"), "{err:#}"); +} + +/// The receipt commits to `sha256(sealed key)`, so a swapped key would be +/// caught — but Object Lock means it cannot be swapped in the first place. +/// Both halves are worth pinning: the retention that prevents it, and the +/// binding that would catch it if retention were ever misconfigured. +#[tokio::test] +async fn the_sealed_key_cannot_be_substituted() { + let store = Store::new(); + let booted = boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .expect("genesis"); + + let (_, other) = StaticKey::new(MasterSecret::from_bytes([77u8; 32])) + .mint() + .await + .unwrap(); + let key_object = format!("origin/{}.key", hex::encode(FS_ID)); + + assert!( + store + .roots + .put_blob(PutBlobInput::new( + &key_object, + other.as_bytes().to_vec().into(), + )) + .await + .is_err(), + "COMPLIANCE retention must refuse to overwrite the sealed key" + ); + + // Unchanged, so the state is still the state the receipt names. + assert_eq!( + boot(&store, &TestNsm::running(0xaa, GUEST_A)) + .await + .unwrap() + .state_root, + booted.state_root + ); +} diff --git a/runtime/tests/guest_logs.rs b/runtime/tests/guest_logs.rs new file mode 100644 index 0000000..f31d50a --- /dev/null +++ b/runtime/tests/guest_logs.rs @@ -0,0 +1,234 @@ +//! Guest stdout and stderr, end to end: a real component writing real bytes, +//! observed through a test sink rather than by scraping a terminal. +//! +//! The unit tests in `guest_io` cover the framing rules directly. What only an +//! integration test can show is that a guest's `write` actually arrives here — +//! that the WASI streams are wired to this pipeline at all, and that nothing +//! between the guest and the sink reorders, merges or loses what it wrote. +//! +//! Records are read from an in-memory [`GuestLogSink`] and never from +//! formatted output. A test that grepped a terminal would pass just as well +//! with the old `inherit_stdio`, which is exactly the arrangement this change +//! removed. +//! +//! Ignored by default because it needs the guest component built first: +//! +//! ```console +//! $ (cd examples/guest-http && cargo build --release --target wasm32-wasip2) +//! $ cargo test -p enclave-runtime --test guest_logs -- --ignored +//! ``` + +use std::path::PathBuf; +use std::sync::{Arc, Mutex}; + +use bytes::Bytes; +use enclave_runtime::guest_io::{GuestLogRecord, GuestLogSink, GuestStream}; +use enclave_runtime::{GuestEnvironment, HostClock, ServeHandle}; +use http_body_util::{BodyExt, Full}; +use s3fs_core::backend::memory::MemoryBackend; +use s3fs_core::{Config, Fs, MasterSecret}; +use wasmtime_wasi_http::p2::bindings::http::types::Scheme; +use wasmtime_wasi_http::p2::body::HyperOutgoingBody; + +fn component_path() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-http/target/wasm32-wasip2/release/guest-http.wasm") +} + +/// Keeps every record, in arrival order. +#[derive(Default)] +struct MemorySink { + records: Mutex>, +} + +#[async_trait::async_trait] +impl GuestLogSink for MemorySink { + async fn emit(&self, record: GuestLogRecord) { + self.records.lock().expect("sink poisoned").push(record); + } +} + +impl MemorySink { + fn len(&self) -> usize { + self.records.lock().expect("sink poisoned").len() + } + + fn messages(&self, stream: GuestStream) -> Vec { + self.records + .lock() + .expect("sink poisoned") + .iter() + .filter(|r| r.stream == stream) + .map(|r| r.message.clone()) + .collect() + } +} + +async fn handle_with(sink: Arc) -> (ServeHandle, enclave_runtime::GuestLogCollector) { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([7u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("creating the memory-backed filesystem"); + + // Started before the handle exists, so there is no window in which a guest + // could write with nothing consuming it. + let (logs, collector) = enclave_runtime::guest_io::start(sink); + let guest = GuestEnvironment::new( + fs, + Box::new(HostClock), + Arc::new(nitro_nsm::fake::FakeNsm::new()), + &[], + &[], + logs, + ) + .expect("building the guest environment"); + + let bytes = std::fs::read(component_path()).unwrap_or_else(|e| { + panic!( + "reading {}: {e}\nbuild it first: (cd examples/guest-http && \ + cargo build --release --target wasm32-wasip2)", + component_path().display() + ) + }); + let engine = ServeHandle::engine_with_watchdog().expect("engine"); + let handle = ServeHandle::new(&engine, &bytes, guest).expect("preparing the guest"); + (handle, collector) +} + +/// Records a guest wrote per request: five lines on stdout, one on stderr. +const PER_REQUEST: usize = 6; + +/// Wait for asynchronous delivery, with a bound. +/// +/// A guest's last unterminated line is emitted when its stream is dropped, and +/// the stream lives in the dispatch task rather than in this one — so the final +/// request's tail arrives shortly *after* the response body has been read. +/// Waiting for it is not papering over a race: eventual delivery is what this +/// pipeline promises, and asserting before it could only ever be flaky. +async fn settle(sink: &MemorySink, expected: usize) { + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(5); + while sink.len() < expected && std::time::Instant::now() < deadline { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } +} + +/// The return type is what drives inference here; an inline cast does not. +fn empty_body() -> HyperOutgoingBody { + Full::new(Bytes::new()) + .map_err(|e: std::convert::Infallible| match e {}) + .boxed_unsync() +} + +async fn get(handle: &ServeHandle, path: &str) -> u16 { + let req = hyper::Request::builder() + .method("GET") + .uri(format!("http://enclave.test{path}")) + .body(empty_body()) + .expect("well-formed request"); + let resp = handle + .handle(Scheme::Http, req, None) + .await + .expect("guest handled the request"); + let status = resp.status().as_u16(); + // Drained, so the guest's instance is finished with before the records are + // read: the last unterminated line is emitted when its stream drops. + let _ = resp.into_body().collect().await.expect("collecting body"); + status +} + +/// The whole point, in one test: bytes a guest wrote reach the runtime's sink, +/// framed into lines, tagged with the stream they came from. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_guest_write_arrives_as_a_framed_record() { + let sink = Arc::new(MemorySink::default()); + let (handle, collector) = handle_with(sink.clone()).await; + assert_eq!(get(&handle, "/log").await, 200); + + // Ends the instance and drains the queue, so what follows is everything + // the guest produced and not a snapshot part way through. + drop(handle); + settle(&sink, PER_REQUEST).await; + collector.shutdown().await; + + let out = sink.messages(GuestStream::Stdout); + let err = sink.messages(GuestStream::Stderr); + + // Two `write_all`s, one line: the runtime joined them. + assert!(out.contains(&"first line".to_string()), "stdout: {out:?}"); + // A blank line the guest wrote is a record the runtime kept. + assert!(out.contains(&String::new()), "stdout: {out:?}"); + // CRLF normalised, with no stray carriage return left behind. + assert!(out.contains(&"windows".to_string()), "stdout: {out:?}"); + // Bytes that are not UTF-8 neither trapped the guest nor lost the line. + assert!( + out.iter() + .any(|m| m.starts_with("invalid ") && m.ends_with(" bytes") && m.contains('\u{fffd}')), + "stdout: {out:?}" + ); + // The final line had no terminator and was emitted when the stream closed. + assert!( + out.contains(&"no trailing newline".to_string()), + "stdout: {out:?}" + ); + + // The distinction that must survive the whole path. + assert_eq!(err, vec!["on stderr".to_string()], "stderr: {err:?}"); + assert!( + !out.contains(&"on stderr".to_string()), + "stderr leaked into stdout: {out:?}" + ); +} + +/// Nothing the guest writes goes anywhere but the sink. If any of this had +/// still been inherited, these records would be on the test's own stdout and +/// the sink would be short. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn no_guest_output_escapes_to_the_hosts_own_streams() { + let sink = Arc::new(MemorySink::default()); + let (handle, collector) = handle_with(sink.clone()).await; + for _ in 0..3 { + assert_eq!(get(&handle, "/log").await, 200); + } + drop(handle); + settle(&sink, 3 * PER_REQUEST).await; + collector.shutdown().await; + + // Six per request: five stdout lines and one stderr. + let out = sink.messages(GuestStream::Stdout); + let err = sink.messages(GuestStream::Stderr); + assert_eq!(err.len(), 3, "stderr: {err:?}"); + assert_eq!(out.len(), 15, "stdout: {out:?}"); +} + +/// Each request is a fresh instance with fresh streams, so an unterminated +/// line from one request must not be joined to the next one's first line. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn one_requests_partial_line_does_not_join_the_next() { + let sink = Arc::new(MemorySink::default()); + let (handle, collector) = handle_with(sink.clone()).await; + assert_eq!(get(&handle, "/log").await, 200); + assert_eq!(get(&handle, "/log").await, 200); + drop(handle); + settle(&sink, 2 * PER_REQUEST).await; + collector.shutdown().await; + + let out = sink.messages(GuestStream::Stdout); + assert_eq!( + out.iter().filter(|m| *m == "no trailing newline").count(), + 2, + "each request's tail must stand alone: {out:?}" + ); + assert!( + !out.iter().any(|m| m.contains("no trailing newlinefirst")), + "two requests' output was spliced: {out:?}" + ); +} diff --git a/runtime/tests/harness.rs b/runtime/tests/harness.rs new file mode 100644 index 0000000..8c9d0fe --- /dev/null +++ b/runtime/tests/harness.rs @@ -0,0 +1,126 @@ +//! The harness a guest author gets, used the way they would use it. +//! +//! Every test here is what a separate repository — one whose component this +//! runtime will serve — would write against `enclave_runtime::testing`. If any +//! of it needs knowledge that lives only in this repository, the harness is not +//! finished. +//! +//! ```console +//! $ (cd examples/guest-http && cargo build --release --target wasm32-wasip2) +//! $ cargo test -p enclave-runtime --test harness -- --ignored +//! ``` + +use enclave_runtime::testing::Enclave; + +fn guest() -> Vec { + let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-http/target/wasm32-wasip2/release/guest-http.wasm"); + std::fs::read(path).expect("build examples/guest-http for wasm32-wasip2") +} + +/// An FCM-shaped registration token, as a client would present. +const DEVICE: &str = "harness-device:APA91bEnclaveRuntimeTestHarness0123456789"; + +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_guest_author_can_stand_up_an_enclave_and_sign_a_request() { + let enclave = Enclave::builder(guest()).start().await.unwrap(); + + let alice = enclave.enrol().await.unwrap(); + let (status, body) = enclave.signed(&alice, "GET", "/counter", "").await.unwrap(); + assert_eq!(status, 200, "{body}"); + assert_eq!(body.trim(), "1"); + + // And the gate is real: the same route without an assertion is refused. + let (status, _, _) = enclave.request("GET", "/counter", &[], "").await.unwrap(); + assert_eq!( + status, 401, + "the harness served a guest without an assertion" + ); +} + +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn two_passkeys_are_two_tenants() { + let enclave = Enclave::builder(guest()).start().await.unwrap(); + let alice = enclave.enrol().await.unwrap(); + let bob = enclave.enrol().await.unwrap(); + assert_ne!(alice.tenant(), bob.tenant()); + + enclave + .signed(&alice, "POST", "/files/who.txt", "alice") + .await + .unwrap(); + enclave + .signed(&bob, "POST", "/files/who.txt", "bob") + .await + .unwrap(); + + let (_, seen_by_alice) = enclave + .signed(&alice, "GET", "/files/who.txt", "") + .await + .unwrap(); + let (_, seen_by_bob) = enclave + .signed(&bob, "GET", "/files/who.txt", "") + .await + .unwrap(); + assert_eq!(seen_by_alice.trim(), "alice"); + assert_eq!(seen_by_bob.trim(), "bob", "one tenant read another's file"); +} + +/// What a guest author most needs to see: work scheduled by one interaction +/// ran later, and woke its owner, with nothing signed at the moment it did. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_task_that_finishes_wakes_its_owner() { + let enclave = Enclave::builder(guest()) + .background_tasks() + .notify() + .start() + .await + .unwrap(); + + let alice = enclave.enrol().await.unwrap(); + let (status, body) = enclave + .signed(&alice, "POST", "/devices", DEVICE) + .await + .unwrap(); + assert_eq!(status, 200, "{body}"); + + let (status, body) = enclave + .signed(&alice, "POST", "/tasks/job", "work") + .await + .unwrap(); + assert_eq!(status, 202, "{body}"); + + let wake = tokio::time::timeout(std::time::Duration::from_secs(20), async { + loop { + if let Some(wake) = enclave.wakes().into_iter().next() { + return wake; + } + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + } + }) + .await + .expect("the finished task never woke anybody"); + + assert_eq!(wake.category, "task-done"); + assert_eq!(wake.reference.as_deref(), Some("job")); + assert_eq!(wake.token, DEVICE); +} + +/// The measurements a client would pin, available before it connects. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn the_harness_says_what_a_client_would_have_to_pin() { + let bytes = guest(); + let enclave = Enclave::builder(bytes.clone()).start().await.unwrap(); + + assert_eq!( + enclave.pcr16(), + nitro_attestation::guest_pcr(&bytes), + "the harness disagrees with the verifier about the guest" + ); + assert!(!enclave.trust_root().is_empty()); + assert!(enclave.url().starts_with("https://127.0.0.1:")); +} diff --git a/runtime/tests/serve_auth.rs b/runtime/tests/serve_auth.rs new file mode 100644 index 0000000..100dce2 --- /dev/null +++ b/runtime/tests/serve_auth.rs @@ -0,0 +1,1019 @@ +//! The rule, over a real socket: **no assertion, no guest.** +//! +//! `serve_guest.rs` tests what happens once a tenant has been resolved. +//! `auth::gate` tests the verification itself. This is the join between them — +//! a real listener, a real TLS handshake, the real dispatch — because the +//! interesting failure is not in either half but in the wiring: a gate that +//! verifies perfectly and is never consulted protects nothing. +//! +//! ```console +//! $ (cd examples/guest-http && cargo build --release --target wasm32-wasip2) +//! $ cargo test -p enclave-runtime --test serve_auth -- --ignored +//! ``` + +use std::path::PathBuf; +use std::sync::Arc; + +use enclave_runtime::{ + AuthEndpoints, ChallengeStore, FilesystemCredentials, Gate, GuestEnvironment, HostClock, + PoolLimits, ServeConfig, SoftwareAuthenticator, Tenancy, TlsIdentity, +}; +use s3fs_core::backend::memory::MemoryBackend; +use s3fs_core::{Config, Fs, MasterSecret}; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +const RP_ID: &str = "enclave.test"; +const ORIGIN: &str = "https://enclave.test"; + +fn component_path() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-http/target/wasm32-wasip2/release/guest-http.wasm") +} + +/// An NSM that signs for real, echoing back whatever it was asked to bind. +/// +/// The auth exchange is now the only place the runtime attests, and it is the +/// exchange a client uses to identify the enclave before approving anything — +/// so this harness has to produce documents that verify, not canned bytes. +/// +/// Its registers behave like the device's: 0–15 locked from the start, 16 free +/// until the harness measures the guest into it, and documents list only the +/// locked ones. So PCR16 is in a document here for the reason it is in a real +/// one — the guest was measured and the register locked. +#[derive(Debug)] +struct SigningNsm { + chain: nitro_attestation::testing::TestChain, + pcrs: std::sync::Mutex>, + /// Counted, so successive draws differ. A device that returned the same + /// bytes every time would mint one tenant id for every passkey, which is + /// the isolation property quietly inverted. + draws: std::sync::atomic::AtomicU64, +} + +/// What `SigningNsm` reports as PCR0, and what a client here pins. +const PCR0: [u8; 48] = [0x5a; 48]; + +impl SigningNsm { + fn new() -> Self { + SigningNsm { + chain: nitro_attestation::testing::TestChain::new().expect("test chain"), + pcrs: std::sync::Mutex::new( + (0..32) + .map(|i| nitro_nsm::Pcr { + locked: i < 16, + value: if i == 0 { + PCR0.to_vec() + } else { + nitro_nsm::PCR_ZERO.to_vec() + }, + }) + .collect(), + ), + draws: std::sync::atomic::AtomicU64::new(0), + } + } +} + +impl nitro_nsm::Nsm for SigningNsm { + fn get_random(&self, buf: &mut [u8]) -> anyhow::Result<()> { + let draw = self.draws.fetch_add(1, std::sync::atomic::Ordering::SeqCst); + for (i, b) in buf.iter_mut().enumerate() { + *b = (i as u8) + .wrapping_mul(7) + .wrapping_add(3) + .wrapping_add(draw as u8); + } + Ok(()) + } + fn attest(&self, request: &nitro_nsm::AttestationRequest) -> anyhow::Result> { + let pcrs = self + .pcrs + .lock() + .unwrap() + .iter() + .enumerate() + .filter(|(_, pcr)| pcr.locked) + .map(|(index, pcr)| (index as u32, pcr.value.clone())) + .collect(); + self.chain + .document_with_pcrs(request.user_data.clone(), request.nonce.clone(), pcrs) + } + fn describe_pcr(&self, index: u16) -> anyhow::Result { + self.pcrs + .lock() + .unwrap() + .get(index as usize) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no PCR{index}")) + } + fn extend_pcr(&self, index: u16, data: &[u8]) -> anyhow::Result> { + let mut pcrs = self.pcrs.lock().unwrap(); + let pcr = pcrs + .get_mut(index as usize) + .ok_or_else(|| anyhow::anyhow!("no PCR{index}"))?; + anyhow::ensure!(!pcr.locked, "PCR{index} is read-only"); + pcr.value = nitro_nsm::pcr_extend(&pcr.value, data); + Ok(pcr.value.clone()) + } + fn lock_pcr(&self, index: u16) -> anyhow::Result<()> { + let mut pcrs = self.pcrs.lock().unwrap(); + let pcr = pcrs + .get_mut(index as usize) + .ok_or_else(|| anyhow::anyhow!("no PCR{index}"))?; + anyhow::ensure!(!pcr.locked, "PCR{index} is read-only"); + pcr.locked = true; + Ok(()) + } + fn describe(&self) -> String { + "signing test NSM".into() + } +} + +struct Harness { + addr: std::net::SocketAddr, + nsm: Arc, + guest_bytes: Vec, +} + +async fn start() -> Harness { + start_with_background(false).await +} + +async fn start_with_background(background: bool) -> Harness { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([9u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("filesystem"); + + let bytes = std::fs::read(component_path()).unwrap_or_else(|e| { + panic!( + "reading {}: {e}\nbuild it first: (cd examples/guest-http && \ + cargo build --release --target wasm32-wasip2)", + component_path().display() + ) + }); + let bytes_for_harness = bytes.clone(); + + let nsm = Arc::new(SigningNsm::new()); + // As `main` does, before anything could attest: the documents this server + // produces carry PCR16 because the guest it serves was measured into it. + enclave_runtime::measure_guest(nsm.as_ref(), &bytes).expect("measuring the guest"); + let entropy: Arc = nsm.clone(); + let credentials = Arc::new(FilesystemCredentials::new(fs.clone())); + let gate = Arc::new(Gate::new( + enclave_runtime::build_relying_party(RP_ID, ORIGIN, &[]).expect("relying party"), + ChallengeStore::new(std::time::Duration::from_secs(60), 256), + credentials.clone(), + enclave_runtime::TokenStore::new( + std::time::Duration::from_secs(60), + enclave_runtime::DEFAULT_TOKEN_CAPACITY, + ), + )); + let auth = Arc::new(AuthEndpoints::new( + gate.clone(), + credentials, + fs.clone(), + entropy.clone(), + )); + + // Detached deliberately: the collector runs for as long as this + // environment can send, which is what a test wants. Production drains it + // explicitly instead. + let (logs, _collector) = + enclave_runtime::guest_io::start(std::sync::Arc::new(enclave_runtime::TracingLogSink)); + let guest = GuestEnvironment::new(fs, Box::new(HostClock), entropy, &[], &[], logs) + .expect("guest environment"); + let tls = TlsIdentity::self_signed(&[RP_ID.to_string()]).expect("tls identity"); + + let listener = std::net::TcpListener::bind("127.0.0.1:0").expect("bind"); + let addr = listener.local_addr().unwrap(); + drop(listener); + + let nsm_for_server: Arc = nsm.clone(); + tokio::spawn(async move { + let _ = enclave_runtime::serve_component( + &bytes, + guest, + ServeConfig { + background_tasks: background.then(Default::default), + notify: None, + addr, + certificate: Some(enclave_runtime::CertificateSlot::fixed(Arc::new(tls))), + acme: None, + attestation: Some(nsm_for_server), + request_timeout: std::time::Duration::from_secs(30), + max_interaction: std::time::Duration::from_secs(300), + tenancy: Some(Arc::new(Tenancy::new(PoolLimits::default()))), + authentication: Some((auth, gate)), + egress: Default::default(), + }, + ) + .await; + }); + + for _ in 0..400 { + if tokio::net::TcpStream::connect(addr).await.is_ok() { + return Harness { + addr, + nsm, + guest_bytes: bytes_for_harness, + }; + } + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + } + panic!("the server never came up on {addr}"); +} + +/// One request over TLS, with whatever headers the caller wants. +/// A distinct nonce per request, base64url and unpadded. +/// +/// Distinct rather than random: these tests need only that no two requests +/// share one. A real client uses a CSPRNG. +fn raw_nonce() -> Vec { + use std::sync::atomic::{AtomicU64, Ordering}; + static NEXT: AtomicU64 = AtomicU64::new(1_000); + let mut nonce = vec![0x5au8; 20]; + nonce[..8].copy_from_slice(&NEXT.fetch_add(1, Ordering::Relaxed).to_be_bytes()); + nonce +} + +fn fresh_nonce() -> String { + use base64::Engine as _; + use std::sync::atomic::{AtomicU64, Ordering}; + static NEXT: AtomicU64 = AtomicU64::new(1); + let mut nonce = vec![0x5au8; 20]; + nonce[..8].copy_from_slice(&NEXT.fetch_add(1, Ordering::Relaxed).to_be_bytes()); + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(nonce) +} + +/// Like [`https`], but returns the response head and the certificate this +/// connection presented — which is what an attestation binds. +async fn https_full( + addr: std::net::SocketAddr, + method: &str, + path: &str, + nonce: &[u8], + headers: &[(&str, String)], + body: &str, +) -> (u16, String, Vec) { + use base64::Engine as _; + let config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .unwrap() + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAny)) + .with_no_client_auth(); + let connector = tokio_rustls::TlsConnector::from(Arc::new(config)); + let name = rustls::pki_types::ServerName::try_from(RP_ID).unwrap(); + let socket = tokio::net::TcpStream::connect(addr).await.expect("connect"); + let mut stream = connector.connect(name, socket).await.expect("handshake"); + + let mut request = format!( + "{method} {path} HTTP/1.1\r\nHost: {RP_ID}\r\nConnection: close\r\n\ + x-enclave-nonce: {}\r\nContent-Length: {}\r\n", + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(nonce), + body.len() + ); + for (name, value) in headers { + request.push_str(&format!("{name}: {value}\r\n")); + } + request.push_str("\r\n"); + request.push_str(body); + stream.write_all(request.as_bytes()).await.expect("write"); + + let mut raw = Vec::new(); + let _ = stream.read_to_end(&mut raw).await; + let presented = { + let (_, conn) = stream.get_ref(); + conn.peer_certificates() + .and_then(|c| c.first().cloned()) + .expect("server certificate") + .to_vec() + }; + let split = raw + .windows(4) + .position(|w| w == b"\r\n\r\n") + .expect("headers"); + let head = String::from_utf8_lossy(&raw[..split]).to_string(); + let status: u16 = head + .lines() + .next() + .unwrap() + .split_whitespace() + .nth(1) + .unwrap() + .parse() + .unwrap(); + (status, head, presented) +} + +async fn https( + addr: std::net::SocketAddr, + method: &str, + path: &str, + headers: &[(&str, String)], + body: &str, +) -> (u16, String) { + let config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .unwrap() + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAny)) + .with_no_client_auth(); + let connector = tokio_rustls::TlsConnector::from(Arc::new(config)); + let name = rustls::pki_types::ServerName::try_from(RP_ID).unwrap(); + + let socket = tokio::net::TcpStream::connect(addr).await.expect("connect"); + let mut stream = connector.connect(name, socket).await.expect("handshake"); + + // Every request carries a nonce, whether or not the response it gets back + // is attested — the runtime attests `/auth/` exchanges and nothing else, and + // a client behaves the same either way. + let mut request = format!( + "{method} {path} HTTP/1.1\r\nHost: {RP_ID}\r\nConnection: close\r\n\ + x-enclave-nonce: {}\r\nContent-Length: {}\r\n", + fresh_nonce(), + body.len() + ); + for (name, value) in headers { + request.push_str(&format!("{name}: {value}\r\n")); + } + request.push_str("\r\n"); + request.push_str(body); + stream.write_all(request.as_bytes()).await.expect("write"); + + let mut raw = Vec::new(); + let _ = stream.read_to_end(&mut raw).await; + let split = raw + .windows(4) + .position(|w| w == b"\r\n\r\n") + .expect("headers"); + let head = String::from_utf8_lossy(&raw[..split]).to_string(); + let status: u16 = head + .lines() + .next() + .unwrap() + .split_whitespace() + .nth(1) + .unwrap() + .parse() + .unwrap(); + let rest = &raw[split + 4..]; + let body = if head + .to_ascii_lowercase() + .contains("transfer-encoding: chunked") + { + dechunk(rest) + } else { + rest.to_vec() + }; + (status, String::from_utf8_lossy(&body).to_string()) +} + +fn dechunk(body: &[u8]) -> Vec { + let mut out = Vec::new(); + let mut rest = body; + while let Some(end) = rest.windows(2).position(|w| w == b"\r\n") { + let header = String::from_utf8_lossy(&rest[..end]); + let Ok(size) = usize::from_str_radix(header.split(';').next().unwrap_or("").trim(), 16) + else { + break; + }; + rest = &rest[end + 2..]; + if size == 0 || rest.len() < size { + break; + } + out.extend_from_slice(&rest[..size]); + rest = rest.get(size + 2..).unwrap_or(&[]); + } + out +} + +async fn post_json( + addr: std::net::SocketAddr, + path: &str, + value: serde_json::Value, +) -> (u16, serde_json::Value) { + let (status, body) = https( + addr, + "POST", + path, + &[("content-type", "application/json".into())], + &value.to_string(), + ) + .await; + ( + status, + serde_json::from_str(&body).unwrap_or(serde_json::Value::Null), + ) +} + +/// Register a passkey and return it, ready to sign for requests. +async fn enrol(addr: std::net::SocketAddr) -> SoftwareAuthenticator { + let auth = SoftwareAuthenticator::new(RP_ID); + let (status, options) = post_json(addr, "/auth/register/options", serde_json::json!({})).await; + assert_eq!(status, 200, "{options}"); + let challenge = options["options"]["publicKey"]["challenge"] + .as_str() + .expect("a challenge"); + let (status, body) = post_json( + addr, + "/auth/register/verify", + serde_json::json!({ + "registration_id": options["registration_id"], + "credential": auth.register(challenge, ORIGIN), + }), + ) + .await; + assert_eq!(status, 200, "{body}"); + auth +} + +/// A signed request: ask for a challenge bound to it, then send it. +/// Walk the whole flow: challenge, assertion, token, interaction. +/// +/// Three trips, and they have to be three: a passkey is a challenge-response, +/// so the assertion cannot exist until the challenge has been answered, and the +/// token cannot exist until the assertion has been checked. +async fn signed( + addr: std::net::SocketAddr, + auth: &SoftwareAuthenticator, + method: &str, + path: &str, + body: &str, +) -> (u16, String) { + let token = token_for(addr, auth, method, path).await; + https( + addr, + method, + path, + &[("authorization", format!("Bearer {token}"))], + body, + ) + .await +} + +/// Trips one and two, returning the token they produce. +async fn token_for( + addr: std::net::SocketAddr, + auth: &SoftwareAuthenticator, + method: &str, + path: &str, +) -> String { + use base64::Engine as _; + let b64 = base64::engine::general_purpose::URL_SAFE_NO_PAD; + let (path_only, query) = match path.split_once('?') { + Some((p, q)) => (p, Some(q)), + None => (path, None), + }; + + let (status, options) = post_json( + addr, + "/auth/request/options", + serde_json::json!({ + "credential_id": b64.encode(auth.credential_id()), + "method": method, + "path": path_only, + "query": query, + }), + ) + .await; + assert_eq!(status, 200, "{options}"); + + let challenge = options["options"]["publicKey"]["challenge"] + .as_str() + .expect("a challenge"); + let assertion = auth.assert(challenge, ORIGIN); + + let (status, granted) = post_json( + addr, + "/auth/request/verify", + serde_json::json!({ + "challenge_id": options["challenge_id"].as_str().expect("a challenge id"), + "assertion": b64.encode(assertion.to_string()), + }), + ) + .await; + assert_eq!(status, 200, "{granted}"); + granted["token"].as_str().expect("a token").to_string() +} + +/// **The rule.** Nothing reaches the guest without an assertion. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn the_guest_is_unreachable_without_an_assertion() { + let h = start().await; + for path in ["/", "/counter", "/whoami", "/files/x", "/nothing-here"] { + let (status, body) = https(h.addr, "GET", path, &[], "").await; + assert_eq!(status, 401, "{path} reached the guest: {body}"); + // Not the guest's own 404, and not its banner: the guest was never + // called at all. + assert!(!body.contains("guest-http"), "{path} was served: {body}"); + assert!(!body.contains("no route for"), "{path} reached the guest"); + } +} + +/// And with one, it works — so the refusals above mean something. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_signed_request_reaches_the_guest() { + let h = start().await; + let auth = enrol(h.addr).await; + + let (status, body) = signed(h.addr, &auth, "GET", "/counter", "").await; + assert_eq!(status, 200, "{body}"); + assert_eq!(body.trim(), "1"); + + // A second, separately signed request continues the same tenant. + let (status, body) = signed(h.addr, &auth, "GET", "/counter", "").await; + assert_eq!(status, 200, "{body}"); + assert_eq!(body.trim(), "2", "the tenant did not persist"); +} + +/// The guest is told which tenant, and the assertion decided it. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn the_guest_is_told_the_resolved_tenant() { + let h = start().await; + let alice = enrol(h.addr).await; + + let (status, seen) = signed(h.addr, &alice, "GET", "/whoami", "").await; + assert_eq!(status, 200); + assert_eq!(seen.trim().len(), 32, "expected a hex tenant id: {seen}"); + assert_ne!(seen.trim(), "(anonymous)"); +} + +/// **The substitution, over a real socket.** An approval for one body must not +/// **What the approval still names: the interaction.** +/// +/// A token issued for one route cannot be spent on another, so an approval to +/// write one file is not an approval to write a different one. +/// +/// What it deliberately no longer names is the *body* — see +/// `a_token_does_not_bind_the_body` in the gate's own tests. That is the trade +/// this model makes, and it is written down rather than left to be discovered. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_token_cannot_be_moved_to_another_interaction() { + let h = start().await; + let auth = enrol(h.addr).await; + + // Approved for writing one file. + let token = token_for(h.addr, &auth, "POST", "/files/approved.txt").await; + + // Spent on another. + let (status, body) = https( + h.addr, + "POST", + "/files/substituted.txt", + &[("authorization", format!("Bearer {token}"))], + "substituted", + ) + .await; + assert_eq!(status, 401, "a token was moved to another route: {body}"); + + // And the guest genuinely was not called. + let (status, _) = signed(h.addr, &auth, "GET", "/files/substituted.txt", "").await; + assert_eq!( + status, 404, + "the substituted request reached the filesystem" + ); +} + +/// **One approval, one interaction.** A captured token is worthless afterwards. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_token_cannot_be_replayed() { + let h = start().await; + let auth = enrol(h.addr).await; + let token = token_for(h.addr, &auth, "GET", "/counter").await; + let headers = [("authorization", format!("Bearer {token}"))]; + + assert_eq!(https(h.addr, "GET", "/counter", &headers, "").await.0, 200); + let (status, _) = https(h.addr, "GET", "/counter", &headers, "").await; + assert_eq!(status, 401, "a spent token was accepted a second time"); +} + +/// Two passkeys, two tenants, and neither can see the other's data — the whole +/// arrangement, over the wire, from enrollment to storage. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn two_passkeys_are_two_tenants() { + let h = start().await; + let alice = enrol(h.addr).await; + let bob = enrol(h.addr).await; + + let (_, alice_id) = signed(h.addr, &alice, "GET", "/whoami", "").await; + let (_, bob_id) = signed(h.addr, &bob, "GET", "/whoami", "").await; + assert_ne!( + alice_id.trim(), + bob_id.trim(), + "two passkeys resolved to one tenant" + ); + + // The same path, written by each, holds each one's own value. + assert_eq!( + signed(h.addr, &alice, "POST", "/files/secret.txt", "alice's") + .await + .0, + 201 + ); + assert_eq!( + signed(h.addr, &bob, "POST", "/files/secret.txt", "bob's") + .await + .0, + 201 + ); + let (_, seen_by_alice) = signed(h.addr, &alice, "GET", "/files/secret.txt", "").await; + let (_, seen_by_bob) = signed(h.addr, &bob, "GET", "/files/secret.txt", "").await; + assert_eq!(seen_by_alice, "alice's"); + assert_eq!(seen_by_bob, "bob's", "one tenant read another's file"); +} + +/// `/auth/*` answers without an assertion — it has to, or nobody could ever +/// get one — and it performs no cosigner action. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn auth_routes_answer_without_an_assertion() { + let h = start().await; + let (status, _) = post_json(h.addr, "/auth/register/options", serde_json::json!({})).await; + // Answered on its merits, not refused for lack of an assertion: this is the + // one route that must work before any credential exists. + assert_eq!(status, 200); + + let (status, _) = post_json(h.addr, "/auth/nonsense", serde_json::json!({})).await; + assert_eq!(status, 404); +} + +#[derive(Debug)] +struct AcceptAny; + +impl rustls::client::danger::ServerCertVerifier for AcceptAny { + fn verify_server_cert( + &self, + _e: &rustls::pki_types::CertificateDer<'_>, + _i: &[rustls::pki_types::CertificateDer<'_>], + _s: &rustls::pki_types::ServerName<'_>, + _o: &[u8], + _n: rustls::pki_types::UnixTime, + ) -> Result { + Ok(rustls::client::danger::ServerCertVerified::assertion()) + } + + fn verify_tls12_signature( + &self, + _m: &[u8], + _c: &rustls::pki_types::CertificateDer<'_>, + _d: &rustls::DigitallySignedStruct, + ) -> Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + + fn verify_tls13_signature( + &self, + _m: &[u8], + _c: &rustls::pki_types::CertificateDer<'_>, + _d: &rustls::DigitallySignedStruct, + ) -> Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + + fn supported_verify_schemes(&self) -> Vec { + rustls::crypto::aws_lc_rs::default_provider() + .signature_verification_algorithms + .supported_schemes() + } +} + +/// Where the real client binary is, or `None` when it has not been built. +/// +/// Prints the build line and returns `None` rather than failing, because a +/// suite that cannot find it has nothing to say about the client — but a +/// *silent* skip reports green for a test that never ran, so it says so. +fn client_binary() -> Option { + let exe = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../target/release/passkey-client"); + if exe.exists() { + return Some(exe); + } + eprintln!( + "skipping: build it with\n cargo build --release -p enclave-runtime \\\n --features testing --bin passkey-client" + ); + None +} + +/// A passkey file of this test's own. +/// +/// Tagged, not just keyed on the process id: these tests share one process, and +/// one state file between them would mean one test's credential answering +/// another's challenges. +fn client_state(tag: &str) -> PathBuf { + let state = std::env::temp_dir().join(format!("passkey-{}-{tag}.json", std::process::id())); + let _ = std::fs::remove_file(&state); + state +} + +/// One invocation of the real client, as a subprocess, against this harness. +/// +/// Every call is a fresh process that re-reads its passkey, re-runs the whole +/// three-trip ceremony, and re-checks the attestation — which is what a shell +/// script driving this binary does, and the reason it is worth spawning rather +/// than calling in-process. +async fn client(h: &Harness, state: &std::path::Path, args: Vec) -> (bool, String, String) { + let exe = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../target/release/passkey-client"); + let state = state.to_path_buf(); + let pcr16 = hex::encode(nitro_attestation::guest_pcr(&h.guest_bytes)); + let url = format!("https://127.0.0.1:{}", h.addr.port()); + tokio::task::spawn_blocking(move || { + let out = std::process::Command::new(&exe) + .args(["--url", &url, "--state", state.to_str().unwrap()]) + // The harness signs with a `TestChain`, which roots at itself + // rather than AWS. The chain is still verified — this only says + // "do not require the AWS root", which is the whole difference + // between this and production. + .arg("--allow-untrusted-root") + // And which enclave, which the client insists on knowing in both + // halves: `SigningNsm` reports PCR0, and PCR16 holds the guest the + // harness measured into it. + .args(["--pcr0", &hex::encode(PCR0)]) + .args(["--pcr16", &pcr16]) + .args(&args) + .output() + .expect("running passkey-client"); + ( + out.status.success(), + String::from_utf8_lossy(&out.stdout).to_string(), + String::from_utf8_lossy(&out.stderr).to_string(), + ) + }) + .await + .unwrap() +} + +/// The `passkey-client` binary, against a real server — the same path the QEMU +/// harness takes. +/// +/// Worth its own test because the harness cannot be run here (`/dev/vsock` is +/// absent) and a helper that only works in theory would fail at the one moment +/// nobody is watching. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built and the passkey-client binary"] +async fn the_passkey_client_binary_drives_the_gate() { + let h = start().await; + if client_binary().is_none() { + return; + } + let state = client_state("gate"); + let run = |args: Vec| client(&h, &state, args); + + let (ok, _, err) = run(vec!["enrol".into()]).await; + assert!(ok, "enrol failed: {err}"); + + let (ok, body, err) = run(vec!["get".into(), "--path".into(), "/counter".into()]).await; + assert!( + ok, + "a signed request failed: stdout={body:?} stderr={err:?}" + ); + assert_eq!(body.trim(), "1", "unexpected body: {body}"); + + // And the substitution the harness checks: refused, and nothing written. + let (_, out, _) = run(vec![ + "substitute".into(), + "--approved".into(), + "/files/approved.txt".into(), + "--sent".into(), + "/files/substituted.txt".into(), + "--body".into(), + "substituted".into(), + ]) + .await; + assert!( + out.starts_with("401"), + "a token was spent on a route it was not approved for: {out}" + ); + let (ok, _, _) = run(vec![ + "get".into(), + "--path".into(), + "/files/substituted.txt".into(), + ]) + .await; + assert!(!ok, "the substituted request reached the filesystem"); + + let _ = std::fs::remove_file(&state); +} + +/// **Standing authority: one ceremony, and work that runs after it.** +/// +/// Everything else in this file is request and response — the client signs, the +/// guest answers, and the approval is spent by the time the connection closes. +/// Background work is the one place that shape does not hold: the assertion +/// authorises an enqueue, and what it authorised runs later, with nobody there +/// to sign anything at the moment it does. +/// +/// `scheduled_work_requires_authentication_and_runs_for_its_owner` makes that +/// claim with the in-process authenticator. This one makes it with the real +/// binary, out of process, which is the thing a person actually holds — and +/// because every poll below is a fresh process that re-reads its passkey, the +/// credential surviving between invocations is part of what is under test. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built and the passkey-client binary"] +async fn the_passkey_client_binary_schedules_work_that_later_runs_for_it() { + let h = start_with_background(true).await; + if client_binary().is_none() { + return; + } + let state = client_state("tasks"); + let run = |args: Vec| client(&h, &state, args); + + let (ok, _, err) = run(vec!["enrol".into()]).await; + assert!(ok, "enrol failed: {err}"); + + // The approval is spent here, on this route, and never again. + let (ok, out, err) = run(vec![ + "post".into(), + "--path".into(), + "/tasks/client-job".into(), + "--body".into(), + "client work".into(), + ]) + .await; + assert!(ok, "the enqueue was refused: stdout={out:?} stderr={err:?}"); + + tokio::time::timeout(std::time::Duration::from_secs(30), async { + loop { + let (ok, body, err) = run(vec![ + "get".into(), + "--path".into(), + "/tasks/client-job".into(), + ]) + .await; + assert!(ok, "the task record became unreadable: {err}"); + let record: serde_json::Value = serde_json::from_str(&body) + .unwrap_or_else(|e| panic!("not a task record: {e}: {body}")); + match record["status"].as_str() { + Some("completed") => { + // The work that was actually asked for, not merely a + // terminal state. + assert_eq!(record["result"], serde_json::json!(b"client work".to_vec())); + break; + } + Some("failed") => panic!("the scheduled task failed: {body}"), + _ => tokio::time::sleep(std::time::Duration::from_millis(250)).await, + } + } + }) + .await + .expect("the work a signed interaction scheduled never ran"); + + let _ = std::fs::remove_file(&state); +} + +/// **The exchange a client uses to identify the enclave.** +/// +/// `/auth/request/options` is the first trip of three, and the one that goes +/// out on a connection nothing has vouched for yet — so its response is the one +/// that has to carry a document. A client checks it *before* asking a person to +/// approve anything, and then pins the certificate it named for the two trips +/// that follow. +/// +/// This is why there is no separate probe route: the round trip was needed +/// anyway. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn the_challenge_exchange_is_attested_and_binds_its_connection() { + use base64::Engine as _; + let h = start().await; + let auth = enrol(h.addr).await; + + let nonce = raw_nonce(); + let (status, head, presented) = https_full( + h.addr, + "POST", + "/auth/request/options", + &nonce, + &[("content-type", "application/json".to_string())], + &serde_json::json!({ + "credential_id": base64::engine::general_purpose::URL_SAFE_NO_PAD + .encode(auth.credential_id()), + "method": "GET", + "path": "/counter", + }) + .to_string(), + ) + .await; + assert_eq!(status, 200, "{head}"); + + let document = head + .lines() + .find(|l| l.to_ascii_lowercase().starts_with("x-enclave-attestation:")) + .and_then(|l| l.split_once(':')) + .map(|(_, v)| v.trim()) + .expect("the challenge exchange carried no attestation"); + let cose = base64::engine::general_purpose::STANDARD + .decode(document) + .expect("the document is base64"); + + nitro_attestation::verify( + &cose, + &nitro_attestation::VerifyOptions { + trust_root: h.nsm.chain.root_der().to_vec(), + now: std::time::SystemTime::now(), + allow_untrusted_root: false, + }, + ) + .expect("the document verifies") + .expect( + &nitro_attestation::Expectations { + nonce: Some(nonce.clone()), + user_data: Some( + nitro_attestation::AttestationHashes::new(&presented, &h.guest_bytes).serialize(), + ), + ..Default::default() + }, + std::time::SystemTime::now(), + ) + .expect("the document must bind this connection's certificate and this nonce"); +} + +/// And the interaction itself is *not* attested, deliberately. +/// +/// A second document would establish nothing the first has not: the client +/// pinned the certificate the challenge exchange named, and TLS proves the peer +/// holds its private key. What it would cost is an NSM signature per +/// interaction, on a device that is the runtime's throughput ceiling. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_guest_response_is_not_attested() { + let h = start().await; + let auth = enrol(h.addr).await; + let token = token_for(h.addr, &auth, "GET", "/counter").await; + + let nonce = raw_nonce(); + let (status, head, _) = https_full( + h.addr, + "GET", + "/counter", + &nonce, + &[("authorization", format!("Bearer {token}"))], + "", + ) + .await; + assert_eq!(status, 200, "{head}"); + assert!( + !head.to_ascii_lowercase().contains("x-enclave-attestation:"), + "a guest response carried a document:\n{head}" + ); +} + +// Multi-threaded like every other test in this file. With background work +// enabled the scheduler runs the guest on one runtime while this test polls it +// over a socket on the same one; on a current-thread runtime those share a +// single thread, which is a latent flake rather than a design. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "requires built guest-http component"] +async fn scheduled_work_requires_authentication_and_runs_for_its_owner() { + let h = start_with_background(true).await; + assert_eq!( + https(h.addr, "POST", "/tasks/job", &[], "unapproved") + .await + .0, + 401 + ); + let alice = enrol(h.addr).await; + let bob = enrol(h.addr).await; + assert_eq!( + signed(h.addr, &alice, "POST", "/tasks/job", "alice work") + .await + .0, + 202 + ); + assert_eq!(signed(h.addr, &bob, "GET", "/tasks/job", "").await.0, 404); + assert_eq!( + signed(h.addr, &bob, "DELETE", "/tasks/job", "").await.0, + 400 + ); + tokio::time::timeout(std::time::Duration::from_secs(10), async { + loop { + let (status, body) = signed(h.addr, &alice, "GET", "/tasks/job", "").await; + assert_eq!(status, 200, "{body}"); + let record: serde_json::Value = serde_json::from_str(&body).unwrap(); + if record["status"] == "completed" { + assert_eq!(record["result"], serde_json::json!(b"alice work".to_vec())); + break; + } + assert_ne!(record["status"], "failed", "{body}"); + tokio::time::sleep(std::time::Duration::from_millis(20)).await; + } + }) + .await + .expect("scheduled task did not finish"); +} diff --git a/runtime/tests/serve_grpc.rs b/runtime/tests/serve_grpc.rs new file mode 100644 index 0000000..0152cbb --- /dev/null +++ b/runtime/tests/serve_grpc.rs @@ -0,0 +1,792 @@ +//! Bidirectional gRPC streaming, through the runtime, to a Wasm guest. +//! +//! The claim under test is not "a body streams" — `serve_guest` already pins +//! that — but that **both** directions are open at once: the guest answers +//! while the client is still sending, and the client sends more after reading +//! an answer. A runtime that buffered either half would pass a naive echo test +//! and fail every one of these. +//! +//! These drive `ServeHandle::handle` directly rather than over a socket. That +//! is not a shortcut: it is the only way to hold the request body open and feed +//! it one frame at a time while reading the response, which is exactly the +//! interleaving being asserted. `serve_h2` covers the wire. +//! +//! ```console +//! $ (cd examples/guest-grpc && cargo build --release --target wasm32-wasip2) +//! $ cargo test -p enclave-runtime --test serve_grpc -- --include-ignored +//! ``` + +use std::path::PathBuf; +use std::sync::Arc; +use std::time::Duration; + +use bytes::{BufMut, Bytes, BytesMut}; +use enclave_runtime::{GuestEnvironment, HostClock, ServeHandle}; +use http_body_util::BodyExt; +use s3fs_core::backend::memory::MemoryBackend; +use s3fs_core::{Config, Fs, MasterSecret}; +use wasmtime_wasi_http::p2::bindings::http::types::Scheme; + +/// One gRPC frame on its way to the guest, or the error that ended the body. +type ClientFrame = + Result, wasmtime_wasi_http::p2::bindings::http::types::ErrorCode>; +/// The request body the test keeps feeding, one frame at a time. +type OpenBody = http_body_util::StreamBody>; + +fn component_path() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-grpc/target/wasm32-wasip2/release/guest-grpc.wasm") +} + +const SIGN: &str = "/enclave.cosign.v1.SigningSession/Sign"; +const REFUSE: &str = "/enclave.cosign.v1.SigningSession/Refuse"; +const SPIN: &str = "/enclave.cosign.v1.SigningSession/Spin"; +const FIREHOSE: &str = "/enclave.cosign.v1.SigningSession/Firehose"; + +// --- the wire format, host side -------------------------------------------- +// +// Deliberately written out rather than shared with the guest: a test that +// encodes with the same code the guest decodes with proves the two agree with +// themselves, not that either speaks gRPC. + +fn frame(message: &[u8]) -> Bytes { + let mut out = BytesMut::with_capacity(5 + message.len()); + out.put_u8(0); + out.put_u32(message.len() as u32); + out.put_slice(message); + out.freeze() +} + +/// Protobuf for `ClientMsg { session_id, seq, kind, payload }`, by hand. +fn client_msg(session: &str, seq: u64, payload: &[u8]) -> Bytes { + let mut buf = BytesMut::new(); + // field 1, wire type 2 (length-delimited) + buf.put_u8(0x0a); + buf.put_u8(session.len() as u8); + buf.put_slice(session.as_bytes()); + // field 2, wire type 0 (varint) + buf.put_u8(0x10); + let mut n = seq; + loop { + let byte = (n & 0x7f) as u8; + n >>= 7; + if n == 0 { + buf.put_u8(byte); + break; + } + buf.put_u8(byte | 0x80); + } + // field 3, wire type 0: Kind::Round + buf.put_u8(0x18); + buf.put_u8(2); + // field 4, wire type 2 + buf.put_u8(0x22); + buf.put_u8(payload.len() as u8); + buf.put_slice(payload); + frame(&buf.freeze()) +} + +/// Pull `payload` (field 4) out of a `ServerMsg`, which is all these assert on. +fn server_payload(message: &[u8]) -> Vec { + let mut i = 0usize; + while i < message.len() { + let tag = message[i]; + i += 1; + match tag { + // field 1 / 4, length-delimited + 0x0a | 0x22 => { + let len = message[i] as usize; + i += 1; + let value = message[i..i + len].to_vec(); + i += len; + if tag == 0x22 { + return value; + } + } + // varint fields + 0x10 | 0x18 => { + while message[i] & 0x80 != 0 { + i += 1; + } + i += 1; + } + _ => break, + } + } + Vec::new() +} + +/// Reassembles gRPC frames from however the response arrives. +#[derive(Default)] +struct Deframer { + buffer: BytesMut, +} + +impl Deframer { + fn push(&mut self, data: &[u8]) { + self.buffer.extend_from_slice(data); + } + fn next(&mut self) -> Option { + if self.buffer.len() < 5 { + return None; + } + let len = u32::from_be_bytes(self.buffer[1..5].try_into().unwrap()) as usize; + if self.buffer.len() < 5 + len { + return None; + } + let _ = self.buffer.split_to(5); + Some(self.buffer.split_to(len).freeze()) + } +} + +// --- harness ---------------------------------------------------------------- + +async fn grpc_handle() -> ServeHandle { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([3u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("creating the filesystem"); + + let (logs, _collector) = + enclave_runtime::guest_io::start(Arc::new(enclave_runtime::TracingLogSink)); + let guest = GuestEnvironment::new( + fs, + Box::new(HostClock), + Arc::new(nitro_nsm::fake::FakeNsm::new()), + &[], + &[], + logs, + ) + .expect("guest environment"); + + let bytes = std::fs::read(component_path()).unwrap_or_else(|e| { + panic!( + "reading {}: {e}\nbuild it first: (cd examples/guest-grpc && \ + cargo build --release --target wasm32-wasip2)", + component_path().display() + ) + }); + + let engine = ServeHandle::engine_with_watchdog().expect("engine"); + ServeHandle::new(&engine, &bytes, guest).expect("preparing the guest") +} + +/// A request whose body the test keeps feeding. +/// +/// This is what makes the interleaving observable: the call is made with the +/// request body still open, so anything the guest answers is an answer given +/// *before* the client finished asking. +fn open_request( + path: &str, +) -> ( + hyper::Request, + tokio::sync::mpsc::Sender, +) { + let (tx, rx) = tokio::sync::mpsc::channel(4); + let body = http_body_util::StreamBody::new(tokio_stream::wrappers::ReceiverStream::new(rx)); + let req = hyper::Request::builder() + .method("POST") + .uri(format!("http://enclave.test{path}")) + .header("content-type", "application/grpc+proto") + .header("te", "trailers") + .body(body) + .expect("well-formed request"); + (req, tx) +} + +async fn send(tx: &tokio::sync::mpsc::Sender, bytes: Bytes) { + tx.send(Ok(hyper::body::Frame::data(bytes))) + .await + .expect("the guest stopped reading"); +} + +// --- the tests -------------------------------------------------------------- + +/// **The property this whole change exists for.** +/// +/// The guest answers while the request body is still open, and the client +/// sends its next message only after reading that answer. Neither side could +/// make progress if the runtime buffered the other, so the exchange completing +/// at all is the proof. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn both_directions_make_progress_at_once() { + let handle = grpc_handle().await; + let (req, tx) = open_request(SIGN); + + let response = handle + .handle(Scheme::Http, req, None) + .await + .expect("the guest answered"); + assert_eq!(response.status(), 200); + assert_eq!( + response + .headers() + .get("content-type") + .and_then(|v| v.to_str().ok()), + Some("application/grpc+proto") + ); + + let mut body = response.into_body(); + let mut deframer = Deframer::default(); + + for seq in 0..4u64 { + // Ask. + send( + &tx, + client_msg("s-1", seq, format!("round-{seq}").as_bytes()), + ) + .await; + + // And read the answer before asking again. If the runtime were holding + // the request body until it completed, this would block forever. + let answer = loop { + if let Some(message) = deframer.next() { + break message; + } + let frame = tokio::time::timeout(Duration::from_secs(5), body.frame()) + .await + .expect("the guest did not answer while the request was still open") + .expect("the body ended early") + .expect("a frame"); + if let Some(data) = frame.data_ref() { + deframer.push(data); + } + }; + assert_eq!( + server_payload(&answer), + format!("round-{seq}").into_bytes(), + "the guest answered the wrong message" + ); + } + + // Half-close: an orderly end, and the trailers say so. + drop(tx); + let mut status = None; + while let Some(Ok(frame)) = body.frame().await { + if let Some(trailers) = frame.trailers_ref() { + status = trailers.get("grpc-status").cloned(); + } + } + assert_eq!( + status.as_ref().and_then(|v| v.to_str().ok()), + Some("0"), + "a half-closed stream should end OK" + ); +} + +/// A stream stays healthy far past the head timeout. +/// +/// The client paces itself at 200ms a message for well over a second, against +/// a runtime configured to give a response head 400ms. Nothing here is +/// unhealthy — every message is answered — and until the watchdog learned to +/// ask whether bytes were moving rather than how long the call had run, this +/// was the case it could not tell from a runaway. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_stream_outlives_the_request_timeout() { + let handle = grpc_handle().await.with_timeout(Duration::from_millis(400)); + let (req, tx) = open_request(SIGN); + + let response = handle + .handle(Scheme::Http, req, None) + .await + .expect("the guest answered"); + let mut body = response.into_body(); + let mut deframer = Deframer::default(); + + let started = std::time::Instant::now(); + for seq in 0..8u64 { + tokio::time::sleep(Duration::from_millis(200)).await; + send(&tx, client_msg("slow", seq, b"tick")).await; + let answer = loop { + if let Some(message) = deframer.next() { + break message; + } + let frame = tokio::time::timeout(Duration::from_secs(5), body.frame()) + .await + .unwrap_or_else(|_| panic!("the stream stalled at message {seq}")) + .expect("the stream was cut short") + .expect("a frame"); + if let Some(data) = frame.data_ref() { + deframer.push(data); + } + }; + assert_eq!(server_payload(&answer), b"tick".to_vec()); + } + let ran_for = started.elapsed(); + assert!( + ran_for > Duration::from_millis(400), + "the stream did not outlast the timeout, so it proves nothing: {ran_for:?}" + ); +} + +/// And the guarantee that keeps the one above safe. +/// +/// `Spin` answers once and then burns CPU without touching either body. It has +/// no progress to show, so the epoch must still end it — otherwise "a stream +/// may run long" would have quietly become "anything may run forever". +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_spinning_stream_is_still_interrupted() { + let handle = grpc_handle().await.with_timeout(Duration::from_secs(1)); + let (req, tx) = open_request(SPIN); + + let response = handle + .handle(Scheme::Http, req, None) + .await + .expect("the head is set before the guest starts spinning"); + let mut body = response.into_body(); + + send(&tx, client_msg("spin", 0, b"go")).await; + + let started = std::time::Instant::now(); + let outcome = tokio::time::timeout(Duration::from_secs(30), async { + while let Some(frame) = body.frame().await { + if frame.is_err() { + return true; + } + } + true + }) + .await; + let took = started.elapsed(); + + assert!( + outcome.is_ok(), + "a spinning guest was never interrupted: {took:?}" + ); + assert!( + took < Duration::from_secs(25), + "the spinning guest was not stopped promptly: {took:?}" + ); +} + +/// A non-zero gRPC status arrives in the trailers, not the head. +/// +/// This is the shape of every gRPC failure: HTTP 200, and the real answer at +/// the end. A client that read only the head would call this a success. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_refusal_arrives_as_grpc_status_trailers() { + let handle = grpc_handle().await; + let (req, tx) = open_request(REFUSE); + + let response = handle + .handle(Scheme::Http, req, None) + .await + .expect("the guest answered"); + assert_eq!( + response.status(), + 200, + "gRPC reports failure in trailers, never in the status line" + ); + + let mut body = response.into_body(); + send(&tx, client_msg("refused", 0, b"please sign")).await; + drop(tx); + + let mut status = None; + let mut message = None; + while let Some(Ok(frame)) = body.frame().await { + if let Some(trailers) = frame.trailers_ref() { + status = trailers.get("grpc-status").cloned(); + message = trailers.get("grpc-message").cloned(); + } + } + assert_eq!( + status.as_ref().and_then(|v| v.to_str().ok()), + Some("7"), + "the refusal did not reach the client as PERMISSION_DENIED" + ); + assert!( + message + .as_ref() + .and_then(|v| v.to_str().ok()) + .unwrap_or_default() + .contains("not authorized to sign"), + "the refusal carried no reason: {message:?}" + ); +} + +/// A guest writing to a client that is not reading stops, rather than growing. +/// +/// `Firehose` queues 64 KiB-sized messages and never reads. The host's outgoing +/// body holds a couple of chunks and then simply stops polling, so what bounds +/// this is backpressure rather than any limit the guest was asked to respect. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_guest_cannot_outrun_a_client_that_is_not_reading() { + let handle = grpc_handle().await; + let (req, _tx) = open_request(FIREHOSE); + + let response = handle + .handle(Scheme::Http, req, None) + .await + .expect("the guest answered"); + let mut body = response.into_body(); + + // Read one frame, then stop for long enough that an unbounded guest would + // have produced everything it had. + let first = tokio::time::timeout(Duration::from_secs(5), body.frame()) + .await + .expect("the first frame never arrived") + .expect("the body ended early") + .expect("a frame"); + assert!(first.data_ref().is_some(), "expected a data frame first"); + + tokio::time::sleep(Duration::from_millis(300)).await; + + // Then drain the rest. What matters is that it completes: the guest was + // still there to finish, which it could not be if it had run ahead and + // died, and the sleep above did not lose anything. + let mut frames = 1usize; + while let Some(Ok(frame)) = body.frame().await { + if frame.data_ref().is_some() { + frames += 1; + } + } + assert_eq!( + frames, 64, + "the guest lost or duplicated frames while the client was not reading" + ); +} + +/// A tenanted handle, so two clients can be told apart. +async fn tenanted() -> ServeHandle { + let handle = grpc_handle().await; + let tenancy = enclave_runtime::Tenancy::new(enclave_runtime::PoolLimits::default()); + handle.with_tenancy(Arc::new(tenancy)) +} + +/// **The cost of a long stream, stated as a test.** +/// +/// One active guest handler per tenant is the isolation model, not a limit to +/// be raised: a tenant's SQLite database is only safe because exactly one of +/// their requests is ever in flight. A stream is a request, so an open stream +/// occupies that tenant's slot for its whole life and their next request waits. +/// +/// The half that makes it acceptable is the second assertion: another tenant +/// has nothing to queue on and proceeds immediately. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn an_open_stream_holds_its_own_tenant_and_no_other() { + let handle = Arc::new(tenanted().await); + let alice = [0xaa; 16]; + let bob = [0xbb; 16]; + + let (req, tx) = open_request(SIGN); + let response = handle + .handle(Scheme::Http, req, Some(&alice)) + .await + .expect("alice's stream opened"); + let mut body = response.into_body(); + + // Alice's stream is open and answering. + send(&tx, client_msg("alice", 0, b"hello")).await; + let mut deframer = Deframer::default(); + loop { + if deframer.next().is_some() { + break; + } + let frame = tokio::time::timeout(Duration::from_secs(5), body.frame()) + .await + .expect("alice's stream stalled") + .expect("alice's stream ended") + .expect("a frame"); + if let Some(data) = frame.data_ref() { + deframer.push(data); + } + } + + // Alice's *second* request waits behind her stream. + let (blocked, _blocked_tx) = open_request(SIGN); + let alice_again = handle.clone(); + let waiting = tokio::spawn(async move { + alice_again + .handle(Scheme::Http, blocked, Some(&[0xaa; 16])) + .await + .map(|_| ()) + }); + tokio::time::sleep(Duration::from_millis(300)).await; + assert!( + !waiting.is_finished(), + "alice's second request ran while her stream was still open" + ); + + // Bob has nothing to queue on. + let (bobs, bobs_tx) = open_request(SIGN); + let bobs_response = tokio::time::timeout( + Duration::from_secs(5), + handle.handle(Scheme::Http, bobs, Some(&bob)), + ) + .await + .expect("bob waited behind alice, which is the bug this asserts against") + .expect("bob's stream opened"); + assert_eq!(bobs_response.status(), 200); + drop(bobs_tx); + + // And when alice's stream ends, her queued request proceeds. + drop(tx); + while body.frame().await.is_some() {} + assert!( + tokio::time::timeout(Duration::from_secs(10), waiting) + .await + .is_ok(), + "alice's tenant was never released when her stream ended" + ); +} + +/// A stream abandoned mid-flight frees its tenant, and leaves nothing poisoned. +/// +/// Dropping the response body is what a client disconnecting looks like from +/// in here. The guest task is aborted, its instance goes with it, and the next +/// request for that tenant gets a fresh one — which it must, because the +/// interrupted store cannot be re-entered. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn an_abandoned_stream_frees_its_tenant() { + let handle = tenanted().await; + let carol = [0xcc; 16]; + + { + let (req, tx) = open_request(SIGN); + let response = handle + .handle(Scheme::Http, req, Some(&carol)) + .await + .expect("carol's stream opened"); + send(&tx, client_msg("carol", 0, b"hi")).await; + // Both ends dropped without half-closing: the client vanished. + drop(response); + drop(tx); + } + + // The tenant must be usable again, promptly. + let (again, again_tx) = open_request(SIGN); + let response = tokio::time::timeout( + Duration::from_secs(10), + handle.handle(Scheme::Http, again, Some(&carol)), + ) + .await + .expect("the tenant was never released after the stream was abandoned") + .expect("carol's next stream opened"); + assert_eq!(response.status(), 200); + + let mut body = response.into_body(); + let mut deframer = Deframer::default(); + send(&again_tx, client_msg("carol", 1, b"again")).await; + let answer = loop { + if let Some(message) = deframer.next() { + break message; + } + let frame = tokio::time::timeout(Duration::from_secs(5), body.frame()) + .await + .expect("the reused tenant never answered") + .expect("the stream ended early") + .expect("a frame"); + if let Some(data) = frame.data_ref() { + deframer.push(data); + } + }; + assert_eq!( + server_payload(&answer), + b"again".to_vec(), + "the tenant came back but its instance was not usable" + ); +} + +/// Read a stream to its end and return the `grpc-status` it finished with. +async fn status_after_half_close(path: &str, partial: &[u8]) -> (Option, Option) { + let handle = grpc_handle().await; + let (req, tx) = open_request(path); + let response = handle + .handle(Scheme::Http, req, None) + .await + .expect("the guest answered"); + let mut body = response.into_body(); + + send(&tx, Bytes::copy_from_slice(partial)).await; + // Half-close with those bytes stranded: at the HTTP layer this is a clean + // end, which is exactly why the gRPC status has to disagree. + drop(tx); + + let (mut status, mut message) = (None, None); + while let Some(Ok(frame)) = body.frame().await { + if let Some(trailers) = frame.trailers_ref() { + status = trailers + .get("grpc-status") + .and_then(|v| v.to_str().ok()) + .map(str::to_string); + message = trailers + .get("grpc-message") + .and_then(|v| v.to_str().ok()) + .map(str::to_string); + } + } + (status, message) +} + +/// **A truncated header must not read as success.** +/// +/// Four bytes is less than the five-byte prefix, so the guest never saw a +/// length at all. Answering `grpc-status: 0` would tell the client its message +/// had been received when nothing had been. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_stream_ending_on_a_truncated_header_is_not_a_success() { + let (status, message) = status_after_half_close(SIGN, &[0u8, 0, 0, 0]).await; + assert_eq!( + status.as_deref(), + Some("3"), + "a stream cut off inside its header reported success" + ); + assert!( + message.unwrap_or_default().contains("part-way through"), + "the refusal did not say what was wrong" + ); +} + +/// And the subtler shape: a complete header whose payload never finished. +/// +/// The guest read a length of ten and got four bytes. It has been waiting for +/// the rest, and the rest is never coming. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_stream_ending_on_a_truncated_payload_is_not_a_success() { + let mut partial = vec![0u8]; + partial.extend_from_slice(&10u32.to_be_bytes()); + partial.extend_from_slice(b"four"); + + let (status, message) = status_after_half_close(SIGN, &partial).await; + assert_eq!( + status.as_deref(), + Some("3"), + "a stream cut off inside its payload reported success" + ); + assert!(message.unwrap_or_default().contains("part-way through")); +} + +/// The control: a stream that ends between messages still ends cleanly. +/// +/// Without this the two tests above would pass on a guest that simply always +/// refused, which would be a different bug wearing the same trailers. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_stream_ending_between_messages_is_still_a_success() { + let (status, _) = status_after_half_close(SIGN, &client_msg("whole", 0, b"complete")).await; + assert_eq!( + status.as_deref(), + Some("0"), + "a cleanly framed stream was reported as truncated" + ); +} + +/// **The case the epoch watchdog cannot see.** +/// +/// The client opens a stream and then says nothing. The guest is blocked in a +/// host call waiting for a frame, so no wasm executes, so no epoch check is +/// ever reached — the epoch could wait forever and never fire. Meanwhile the +/// stream holds its tenant's only slot. +/// +/// A wall clock is the instrument here, and it works for the same reason the +/// epoch does not: a guest parked in a host call *is* at an await point, so an +/// abort reaches it. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_silent_stream_is_ended_at_its_deadline_and_frees_its_tenant() { + let handle = tenanted() + .await + .with_max_interaction(Duration::from_millis(400)); + let quiet = [0xd0; 16]; + + let (req, tx) = open_request(SIGN); + let response = handle + .handle(Scheme::Http, req, Some(&quiet)) + .await + .expect("the stream opened"); + let mut body = response.into_body(); + + // Not one frame, ever. The sender is held so the request body stays open — + // this is a client that connected and then went quiet, not one that left. + let started = std::time::Instant::now(); + let ended = tokio::time::timeout(Duration::from_secs(20), async { + while let Some(frame) = body.frame().await { + if frame.is_err() { + break; + } + } + }) + .await; + let took = started.elapsed(); + + assert!( + ended.is_ok(), + "a silent stream was never ended: still running after {took:?}" + ); + assert!( + took >= Duration::from_millis(400), + "the stream ended before its deadline, so this proves nothing: {took:?}" + ); + + // And the tenant is usable again — which is the point of ending it. + let (again, again_tx) = open_request(SIGN); + let response = tokio::time::timeout( + Duration::from_secs(10), + handle.handle(Scheme::Http, again, Some(&quiet)), + ) + .await + .expect("the deadline did not free the tenant") + .expect("the tenant's next interaction opened"); + assert_eq!(response.status(), 200); + drop(again_tx); + drop(tx); +} + +/// The other half: a stream that *is* talking runs past the same deadline +/// without being touched, because the deadline bounds neglect rather than +/// duration. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_busy_stream_is_not_ended_by_the_head_timeout() { + let handle = grpc_handle() + .await + .with_timeout(Duration::from_millis(300)) + .with_max_interaction(Duration::from_secs(30)); + let (req, tx) = open_request(SIGN); + + let response = handle + .handle(Scheme::Http, req, None) + .await + .expect("the stream opened"); + let mut body = response.into_body(); + let mut deframer = Deframer::default(); + + let started = std::time::Instant::now(); + for seq in 0..5u64 { + tokio::time::sleep(Duration::from_millis(150)).await; + send(&tx, client_msg("busy", seq, b"tick")).await; + loop { + if deframer.next().is_some() { + break; + } + let frame = tokio::time::timeout(Duration::from_secs(5), body.frame()) + .await + .unwrap_or_else(|_| panic!("the stream stalled at message {seq}")) + .expect("the stream was cut short") + .expect("a frame"); + if let Some(data) = frame.data_ref() { + deframer.push(data); + } + } + } + assert!( + started.elapsed() > Duration::from_millis(300), + "the exchange finished inside the head timeout, so it proves nothing" + ); +} diff --git a/runtime/tests/serve_grpc_wire.rs b/runtime/tests/serve_grpc_wire.rs new file mode 100644 index 0000000..b56b1f5 --- /dev/null +++ b/runtime/tests/serve_grpc_wire.rs @@ -0,0 +1,538 @@ +//! The same guest, driven by a real gRPC client over a real TLS connection. +//! +//! `serve_grpc` pins the streaming semantics through the dispatch path, where +//! the request body can be held open a frame at a time. What it cannot show is +//! that the bytes on the wire are *gRPC* rather than a private framing both +//! halves of this repository happen to agree on — the guest hand-frames gRPC +//! because `tonic` does not build for `wasm32-wasip2`, so checking it against +//! an implementation that came from somewhere else is the point. +//! +//! `tonic` is the client here, over HTTP/2 negotiated by ALPN, and it decodes +//! with `prost` rather than with anything from the guest. +//! +//! ```console +//! $ (cd examples/guest-grpc && cargo build --release --target wasm32-wasip2) +//! $ cargo test -p enclave-runtime --test serve_grpc_wire -- --include-ignored +//! ``` + +use std::path::PathBuf; +use std::sync::{Arc, Mutex}; + +use enclave_runtime::{GuestEnvironment, HostClock, ServeConfig, TlsIdentity}; +use nitro_attestation::testing::TestChain; +use nitro_nsm::{AttestationRequest, Nsm}; +use s3fs_core::backend::memory::MemoryBackend; +use s3fs_core::{Config, Fs, MasterSecret}; + +fn component_path() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-grpc/target/wasm32-wasip2/release/guest-grpc.wasm") +} + +// --- the messages, decoded by prost on this side ---------------------------- + +#[derive(Clone, PartialEq, prost::Message)] +struct ClientMsg { + #[prost(string, tag = "1")] + session_id: String, + #[prost(uint64, tag = "2")] + seq: u64, + #[prost(int32, tag = "3")] + kind: i32, + #[prost(bytes = "vec", tag = "4")] + payload: Vec, +} + +#[derive(Clone, PartialEq, prost::Message)] +struct ServerMsg { + #[prost(string, tag = "1")] + session_id: String, + #[prost(uint64, tag = "2")] + seq: u64, + #[prost(int32, tag = "3")] + kind: i32, + #[prost(bytes = "vec", tag = "4")] + payload: Vec, +} + +// --- an NSM that signs what it was actually asked to bind -------------------- + +#[derive(Debug)] +struct SigningNsm { + chain: TestChain, + pcr0: [u8; 48], + last: Mutex>, +} + +impl SigningNsm { + fn new() -> Self { + SigningNsm { + chain: TestChain::new().expect("test chain"), + pcr0: [0x5a; 48], + last: Mutex::new(None), + } + } +} + +impl Nsm for SigningNsm { + fn get_random(&self, buf: &mut [u8]) -> anyhow::Result<()> { + for (i, b) in buf.iter_mut().enumerate() { + *b = (i as u8).wrapping_mul(7).wrapping_add(3); + } + Ok(()) + } + fn attest(&self, request: &AttestationRequest) -> anyhow::Result> { + *self.last.lock().unwrap() = Some(request.clone()); + self.chain + .document(request.user_data.clone(), request.nonce.clone(), self.pcr0) + } + fn describe_pcr(&self, index: u16) -> anyhow::Result { + Ok(nitro_nsm::Pcr { + locked: index < 3, + value: if index == 0 { + self.pcr0.to_vec() + } else { + nitro_nsm::PCR_ZERO.to_vec() + }, + }) + } + fn extend_pcr(&self, _index: u16, _data: &[u8]) -> anyhow::Result> { + anyhow::bail!("this fake does not model PCR extension") + } + fn lock_pcr(&self, _index: u16) -> anyhow::Result<()> { + anyhow::bail!("this fake does not model PCR locking") + } + fn describe(&self) -> String { + "signing test NSM".into() + } +} + +#[derive(Debug)] +struct AcceptAny; + +impl rustls::client::danger::ServerCertVerifier for AcceptAny { + fn verify_server_cert( + &self, + _e: &rustls::pki_types::CertificateDer<'_>, + _i: &[rustls::pki_types::CertificateDer<'_>], + _n: &rustls::pki_types::ServerName<'_>, + _o: &[u8], + _t: rustls::pki_types::UnixTime, + ) -> Result { + Ok(rustls::client::danger::ServerCertVerified::assertion()) + } + fn verify_tls12_signature( + &self, + _m: &[u8], + _c: &rustls::pki_types::CertificateDer<'_>, + _d: &rustls::DigitallySignedStruct, + ) -> Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + fn verify_tls13_signature( + &self, + _m: &[u8], + _c: &rustls::pki_types::CertificateDer<'_>, + _d: &rustls::DigitallySignedStruct, + ) -> Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + fn supported_verify_schemes(&self) -> Vec { + rustls::crypto::aws_lc_rs::default_provider() + .signature_verification_algorithms + .supported_schemes() + } +} + +struct Harness { + addr: std::net::SocketAddr, +} + +async fn start() -> Harness { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([11u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("creating the filesystem"); + + let nsm = Arc::new(SigningNsm::new()); + let guest_bytes = std::fs::read(component_path()).unwrap_or_else(|e| { + panic!( + "reading {}: {e}\nbuild it first: (cd examples/guest-grpc && \ + cargo build --release --target wasm32-wasip2)", + component_path().display() + ) + }); + + let (logs, _collector) = + enclave_runtime::guest_io::start(Arc::new(enclave_runtime::TracingLogSink)); + let guest = GuestEnvironment::new(fs, Box::new(HostClock), nsm.clone(), &[], &[], logs) + .expect("guest environment"); + + let identity = + Arc::new(TlsIdentity::self_signed(&["enclave.test".to_string()]).expect("tls identity")); + + let listener = std::net::TcpListener::bind("127.0.0.1:0").expect("bind"); + let addr = listener.local_addr().unwrap(); + drop(listener); + + let bytes = guest_bytes.clone(); + let nsm_for_server = nsm.clone(); + tokio::spawn(async move { + let _ = enclave_runtime::serve_component( + &bytes, + guest, + ServeConfig { + background_tasks: None, + notify: None, + addr, + certificate: Some(enclave_runtime::CertificateSlot::fixed(identity)), + acme: None, + attestation: Some(nsm_for_server), + request_timeout: std::time::Duration::from_secs(30), + max_interaction: std::time::Duration::from_secs(300), + tenancy: None, + authentication: None, + egress: Default::default(), + }, + ) + .await; + }); + + let mut ready = false; + for _ in 0..400 { + if tokio::net::TcpStream::connect(addr).await.is_ok() { + ready = true; + break; + } + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + } + assert!(ready, "the server never came up on {addr}"); + + Harness { addr } +} + +/// Adapts hyper's h2 connection to the `tower::Service` tonic expects. +/// +/// tonic's own transport would build its own connection; this one is already +/// open, so the test can hold the certificate that was presented on it and +/// check the attestation against that exact leaf. +#[derive(Clone)] +struct H2Service(hyper::client::conn::http2::SendRequest); + +impl tower::Service> for H2Service { + type Response = http::Response; + type Error = hyper::Error; + type Future = std::pin::Pin< + Box> + Send>, + >; + + fn poll_ready( + &mut self, + cx: &mut std::task::Context<'_>, + ) -> std::task::Poll> { + self.0.poll_ready(cx) + } + + fn call(&mut self, req: http::Request) -> Self::Future { + let mut sender = self.0.clone(); + Box::pin(async move { sender.send_request(req).await }) + } +} + +fn nonce() -> Vec { + use std::sync::atomic::{AtomicU64, Ordering}; + static NEXT: AtomicU64 = AtomicU64::new(1); + let mut n = vec![0x2bu8; 20]; + n[..8].copy_from_slice(&NEXT.fetch_add(1, Ordering::Relaxed).to_be_bytes()); + n +} + +fn nonce_header(nonce: &[u8]) -> String { + use base64::Engine as _; + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(nonce) +} + +/// **A real gRPC client driving a real bidirectional stream.** +/// +/// tonic, over TLS and ALPN-negotiated HTTP/2, decoding with prost — so the +/// guest's hand-written framing is checked against an implementation that did +/// not come from this repository. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn tonic_drives_a_bidirectional_stream_over_the_wire() { + let harness = start().await; + let chosen_nonce = nonce(); + + // One TLS connection, whose certificate this test keeps. + let mut config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .unwrap() + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAny)) + .with_no_client_auth(); + config.alpn_protocols = vec![b"h2".to_vec()]; + + let connector = tokio_rustls::TlsConnector::from(Arc::new(config)); + let name = rustls::pki_types::ServerName::try_from("enclave.test").unwrap(); + let socket = tokio::net::TcpStream::connect(harness.addr) + .await + .expect("connect"); + let stream = connector.connect(name, socket).await.expect("handshake"); + let presented = { + let (_, conn) = stream.get_ref(); + assert_eq!( + conn.alpn_protocol(), + Some(&b"h2"[..]), + "gRPC cannot run without h2" + ); + conn.peer_certificates() + .and_then(|c| c.first().cloned()) + .expect("server certificate") + .to_vec() + }; + + let (sender, connection) = hyper::client::conn::http2::handshake( + hyper_util::rt::TokioExecutor::new(), + hyper_util::rt::TokioIo::new(stream), + ) + .await + .expect("h2 handshake"); + tokio::spawn(async move { + let _ = connection.await; + }); + + let mut client = tonic::client::Grpc::with_origin( + H2Service(sender), + "https://enclave.test".parse().expect("origin"), + ); + client.ready().await.expect("client ready"); + + // The client controls its own send stream, so it can hold back until the + // attestation checks out. + let (tx, rx) = tokio::sync::mpsc::channel::(4); + let mut request = tonic::Request::new(tokio_stream::wrappers::ReceiverStream::new(rx)); + request.metadata_mut().insert( + "x-enclave-nonce", + nonce_header(&chosen_nonce).parse().expect("nonce header"), + ); + + let response = client + .streaming( + request, + "/enclave.cosign.v1.SigningSession/Sign" + .parse() + .expect("path"), + tonic_prost::ProstCodec::::default(), + ) + .await + .expect("the stream opened"); + + // The attestation is *not* checked here, and that is the design rather than + // an omission: the runtime attests the `/auth/` exchange, where a client + // identifies the enclave before approving anything, and the interaction + // itself is pinned to the certificate that exchange named. `serve_auth` + // covers that half. What this test is for is tonic on the wire. + let _ = &presented; + + // --- and only now, talk ------------------------------------------------- + let mut inbound = response.into_inner(); + for seq in 0..4u64 { + tx.send(ClientMsg { + session_id: "wire".into(), + seq, + kind: 2, + payload: format!("round-{seq}").into_bytes(), + }) + .await + .expect("the guest stopped reading"); + + let reply = tokio::time::timeout(std::time::Duration::from_secs(5), inbound.message()) + .await + .expect("the guest did not answer while the request stream was open") + .expect("a message") + .expect("the stream ended early"); + assert_eq!(reply.seq, seq); + assert_eq!( + reply.payload, + format!("round-{seq}").into_bytes(), + "tonic decoded something the guest did not send" + ); + } + + // Half-close, and read the status a real client reads. + drop(tx); + assert!( + inbound.message().await.expect("clean end").is_none(), + "the guest kept talking after the client half-closed" + ); + let trailers = inbound.trailers().await.expect("trailers"); + assert_eq!( + trailers + .as_ref() + .and_then(|t| t.get("grpc-status")) + .and_then(|v| v.to_str().ok()), + Some("0"), + "a half-closed stream should end OK" + ); +} + +/// Opening a connection for tonic, since every test past the first wants one. +async fn grpc_client(addr: std::net::SocketAddr) -> tonic::client::Grpc { + let mut config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .unwrap() + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAny)) + .with_no_client_auth(); + config.alpn_protocols = vec![b"h2".to_vec()]; + + let connector = tokio_rustls::TlsConnector::from(Arc::new(config)); + let name = rustls::pki_types::ServerName::try_from("enclave.test").unwrap(); + let socket = tokio::net::TcpStream::connect(addr).await.expect("connect"); + let stream = connector.connect(name, socket).await.expect("handshake"); + let (sender, connection) = hyper::client::conn::http2::handshake( + hyper_util::rt::TokioExecutor::new(), + hyper_util::rt::TokioIo::new(stream), + ) + .await + .expect("h2 handshake"); + tokio::spawn(async move { + let _ = connection.await; + }); + let mut client = tonic::client::Grpc::with_origin( + H2Service(sender), + "https://enclave.test".parse().expect("origin"), + ); + client.ready().await.expect("client ready"); + client +} + +/// A refusal reaches a real client as a gRPC status, not as an HTTP failure. +/// +/// This is the shape of every gRPC error: the head is 200 and the answer is in +/// the trailers. A client reading only the head would call it a success, which +/// is exactly why `tonic` — rather than this test's own parsing — has to be the +/// thing that reports it. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_refusal_reaches_tonic_as_permission_denied() { + let harness = start().await; + let mut client = grpc_client(harness.addr).await; + + let (tx, rx) = tokio::sync::mpsc::channel::(4); + let mut request = tonic::Request::new(tokio_stream::wrappers::ReceiverStream::new(rx)); + request.metadata_mut().insert( + "x-enclave-nonce", + nonce_header(&nonce()).parse().expect("nonce header"), + ); + + let response = client + .streaming( + request, + "/enclave.cosign.v1.SigningSession/Refuse" + .parse() + .expect("path"), + tonic_prost::ProstCodec::::default(), + ) + .await + .expect("the stream opened"); + + let mut inbound = response.into_inner(); + tx.send(ClientMsg { + session_id: "refused".into(), + seq: 0, + kind: 2, + payload: b"please sign".to_vec(), + }) + .await + .expect("the guest stopped reading"); + drop(tx); + + // The one answer it does give arrives normally. + let first = inbound + .message() + .await + .expect("a message") + .expect("the stream ended before answering"); + assert_eq!(first.payload, b"please sign".to_vec()); + + // And then the refusal, which tonic surfaces as a `Status`. + let status = inbound + .message() + .await + .expect_err("the guest refused, so this must not be a clean end"); + assert_eq!( + status.code(), + tonic::Code::PermissionDenied, + "the refusal did not reach the client as a gRPC status: {status:?}" + ); + assert!( + status.message().contains("not authorized to sign"), + "the refusal carried no reason: {status:?}" + ); +} + +/// A gRPC deadline is the guest's business, and the runtime says so by +/// forwarding it untouched. +/// +/// The runtime has deadlines of its own — the head timeout and the progress +/// watchdog — and they are its own precisely so that a client cannot lengthen +/// them by asking. `grpc-timeout` is neither honoured nor stripped: it reaches +/// the guest, which is the only party that knows what its work is worth. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-grpc built for wasm32-wasip2"] +async fn a_grpc_timeout_does_not_shorten_or_lengthen_the_runtime_deadlines() { + let harness = start().await; + let mut client = grpc_client(harness.addr).await; + + let (tx, rx) = tokio::sync::mpsc::channel::(4); + let mut request = tonic::Request::new(tokio_stream::wrappers::ReceiverStream::new(rx)); + request.metadata_mut().insert( + "x-enclave-nonce", + nonce_header(&nonce()).parse().expect("nonce header"), + ); + // A deadline far longer than anything this test takes. If the runtime were + // acting on it the call would still work; what would break is the + // assumption that it is the guest's to interpret. + request + .metadata_mut() + .insert("grpc-timeout", "600S".parse().expect("timeout header")); + + let response = client + .streaming( + request, + "/enclave.cosign.v1.SigningSession/Sign" + .parse() + .expect("path"), + tonic_prost::ProstCodec::::default(), + ) + .await + .expect("a deadline header must not stop the stream opening"); + + let mut inbound = response.into_inner(); + tx.send(ClientMsg { + session_id: "deadline".into(), + seq: 0, + kind: 2, + payload: b"still here".to_vec(), + }) + .await + .expect("the guest stopped reading"); + + let reply = inbound + .message() + .await + .expect("a message") + .expect("the stream ended early"); + assert_eq!(reply.payload, b"still here".to_vec()); +} diff --git a/runtime/tests/serve_guest.rs b/runtime/tests/serve_guest.rs new file mode 100644 index 0000000..36d0d60 --- /dev/null +++ b/runtime/tests/serve_guest.rs @@ -0,0 +1,836 @@ +//! End-to-end for the serving path: a real `wasi:http/proxy` component, +//! dispatched through the real linker, reading and writing a real block store. +//! +//! It runs over [`MemoryBackend`] rather than MinIO, so it needs no Docker and +//! no network — the thing under test is the HTTP dispatch and the guest's view +//! of the filesystem, neither of which cares whether the blocks land in S3 or +//! in a `HashMap`. The S3 backend has its own integration suite. +//! +//! Ignored by default because it needs the guest component built first: +//! +//! ```console +//! $ (cd examples/guest-http && cargo build --release --target wasm32-wasip2) +//! $ cargo test -p enclave-runtime --test serve_guest -- --ignored +//! ``` + +use std::path::PathBuf; +use std::sync::Arc; + +use bytes::Bytes; +use enclave_runtime::{GuestEnvironment, HostClock, ServeHandle}; +use http_body_util::{BodyExt, Full}; +use s3fs_core::backend::memory::MemoryBackend; +use s3fs_core::{Config, Fs, MasterSecret}; +use wasmtime_wasi_http::p2::bindings::http::types::Scheme; +use wasmtime_wasi_http::p2::body::HyperOutgoingBody; + +fn component_path() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-http/target/wasm32-wasip2/release/guest-http.wasm") +} + +fn body(bytes: &[u8]) -> HyperOutgoingBody { + Full::new(Bytes::copy_from_slice(bytes)) + .map_err(|e: std::convert::Infallible| match e {}) + .boxed_unsync() +} + +/// One filesystem, one compiled guest, many requests — the arrangement the +/// runtime actually uses. +async fn handle_for(env: &[(String, String)]) -> ServeHandle { + handle_for_with(env).await +} + +/// A handle over a filesystem that already exists. +/// +/// Separated so a test can build a *second* handle over the first one's +/// storage — the only way to ask what survives a restart. +async fn handle_over_with(fs: Arc, env: &[(String, String)]) -> ServeHandle { + // Detached deliberately: the collector runs for as long as this + // environment can send, which is what a test wants. Production drains it + // explicitly instead. + let (logs, _collector) = + enclave_runtime::guest_io::start(std::sync::Arc::new(enclave_runtime::TracingLogSink)); + let guest = GuestEnvironment::new( + fs, + Box::new(HostClock), + Arc::new(nitro_nsm::fake::FakeNsm::new()), + env, + &[], + logs, + ) + .expect("building the guest environment"); + + let bytes = std::fs::read(component_path()).unwrap_or_else(|e| { + panic!( + "reading {}: {e}\nbuild it first: (cd examples/guest-http && \ + cargo build --release --target wasm32-wasip2)", + component_path().display() + ) + }); + + let engine = ServeHandle::engine_with_watchdog().expect("engine"); + ServeHandle::new(&engine, &bytes, guest).expect("preparing the guest") +} + +/// A runtime filesystem plus a handle. +async fn handle_for_with(env: &[(String, String)]) -> ServeHandle { + let backend = Arc::new(MemoryBackend::new()); + // `create`, not `mount`: mounting stopped formatting an empty store when + // the boot machine landed, because a store that answers "nothing" is now a + // refusal rather than an invitation to make a fresh filesystem. + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([7u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("creating the memory-backed filesystem"); + handle_over_with(fs, env).await +} + +async fn get_as(handle: &ServeHandle, path: &str, tenant: Option<&[u8; 16]>) -> (u16, String) { + request_as(handle, "GET", path, &[], tenant).await +} + +async fn request_as( + handle: &ServeHandle, + method: &str, + path: &str, + body_bytes: &[u8], + tenant: Option<&[u8; 16]>, +) -> (u16, String) { + let req = hyper::Request::builder() + .method(method) + .uri(format!("http://enclave.test{path}")) + .body(body(body_bytes)) + .expect("well-formed request"); + let resp = handle + .handle(Scheme::Http, req, tenant) + .await + .expect("guest handled the request"); + let status = resp.status().as_u16(); + let collected = resp.into_body().collect().await.expect("collecting body"); + ( + status, + String::from_utf8_lossy(&collected.to_bytes()).into_owned(), + ) +} + +async fn get(handle: &ServeHandle, path: &str) -> (u16, String) { + request(handle, "GET", path, &[]).await +} + +async fn request( + handle: &ServeHandle, + method: &str, + path: &str, + body_bytes: &[u8], +) -> (u16, String) { + let req = hyper::Request::builder() + .method(method) + .uri(format!("http://enclave.test{path}")) + .body(body(body_bytes)) + .expect("well-formed request"); + let resp = handle + .handle(Scheme::Http, req, None) + .await + .expect("guest handled the request"); + let status = resp.status().as_u16(); + let collected = resp.into_body().collect().await.expect("collecting body"); + ( + status, + String::from_utf8_lossy(&collected.to_bytes()).into_owned(), + ) +} + +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn the_guest_answers_requests() { + let handle = handle_for(&[]).await; + let (status, body) = get(&handle, "/").await; + assert_eq!(status, 200, "body: {body}"); + assert!(body.contains("guest-http"), "unexpected banner: {body}"); +} + +/// The load-bearing test. Each request gets its own `Store` and its own guest +/// instance; only a *committed* filesystem carries the count between them. If +/// this returns 1 twice, requests are being served against separate or +/// uncommitted views of the store. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn state_written_by_one_request_is_visible_to_the_next() { + let handle = handle_for(&[]).await; + for expected in 1..=3 { + let (status, body) = get(&handle, "/counter").await; + assert_eq!(status, 200, "body: {body}"); + assert_eq!(body.trim(), expected.to_string(), "counter did not advance"); + } +} + +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_file_written_over_http_reads_back() { + let handle = handle_for(&[]).await; + let payload = b"the block store round-trips through wasi:http"; + + let (status, body) = request(&handle, "POST", "/files/note.txt", payload).await; + assert_eq!(status, 201, "body: {body}"); + + let (status, body) = get(&handle, "/files/note.txt").await; + assert_eq!(status, 200); + assert_eq!(body.as_bytes(), payload); +} + +/// The environment policy reaches an HTTP guest by the same path a command +/// guest uses — `GuestEnvironment` builds both, which is why it exists. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn the_environment_policy_reaches_the_guest() { + let handle = handle_for(&[("DEPLOYMENT".to_string(), "test".to_string())]).await; + let (status, body) = get(&handle, "/env").await; + assert_eq!(status, 200); + assert!(body.contains("DEPLOYMENT=test"), "env was: {body}"); +} + +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn an_unknown_route_is_the_guests_own_404() { + let handle = handle_for(&[]).await; + let (status, _) = get(&handle, "/nothing-here").await; + assert_eq!(status, 404); +} + +/// A path escaping the guest's own directory is refused *by the guest*. The +/// preopen it holds is the filesystem root, so nothing below would have +/// stopped it — which is exactly why the example does the check and why this +/// test guards the example. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_traversing_path_is_refused() { + let handle = handle_for(&[]).await; + let (status, body) = get(&handle, "/files/../counter").await; + assert_ne!(status, 200, "traversal must not succeed: {body}"); +} + +/// `wasi:http/outgoing-handler` is linked, so a guest can call it. It must +/// fail. This is the enclave's egress boundary, checked through the same +/// linker the runtime uses rather than against the policy type in isolation. +#[tokio::test(flavor = "multi_thread")] +async fn the_linker_denies_guest_egress() { + use enclave_runtime::EgressPolicy; + use wasmtime_wasi_http::p2::{types::OutgoingRequestConfig, WasiHttpHooks}; + + let mut policy = EgressPolicy::Denied; + let req = hyper::Request::builder() + .uri("https://example.invalid/") + .body(body(b"")) + .unwrap(); + let config = OutgoingRequestConfig { + use_tls: true, + connect_timeout: std::time::Duration::from_secs(1), + first_byte_timeout: std::time::Duration::from_secs(1), + between_bytes_timeout: std::time::Duration::from_secs(1), + }; + let Err(err) = policy.send_request(req, config) else { + panic!("egress must be refused"); + }; + assert!(format!("{err:?}").contains("HttpRequestDenied"), "{err:?}"); +} + +/// A guest that never returns and never answers. +/// +/// Without a watchdog this is not a slow request, it is a permanent one: the +/// `oneshot::Sender` lives in the `Store`, the `Store` was moved into the +/// spawned task, so `receiver.await` can never resolve and the tenant's lock is +/// never released — which would finish that tenant for the life of the process, +/// though not anybody else's. +/// +/// The assertions are ordered accordingly — the second one is the point. The +/// first only shows the request gave up; the second shows the *server* did not. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_hung_guest_does_not_wedge_the_server() { + let handle = handle_for(&[]) + .await + .with_timeout(std::time::Duration::from_secs(3)); + + let req = hyper::Request::builder() + .method("GET") + .uri("http://enclave.test/hang") + .body(body(b"")) + .expect("well-formed request"); + + // Belt and braces: if the watchdog regresses this fails rather than hanging + // the whole suite, which is what it would do otherwise. + let hung = tokio::time::timeout( + std::time::Duration::from_secs(20), + handle.handle(Scheme::Http, req, None), + ) + .await + .expect("the watchdog did not fire; the request hung"); + + // Either mechanism is a correct outcome and which one fires depends on + // where the guest is stuck: spinning wasm is trapped by the epoch, a guest + // parked in a host call is abandoned by the timeout. Asserting on one would + // make this a test of the message rather than of the guarantee. + hung.expect_err("a guest that never answers must not succeed"); + + // The permit was released and the guest instance is gone, so the next + // request is served normally. This is the assertion that matters. + let (status, body) = get(&handle, "/").await; + assert_eq!( + status, 200, + "the server was wedged by the hung request: {body}" + ); +} + +// --------------------------------------------------------------------------- +// One instance per request. No two requests share one. +// --------------------------------------------------------------------------- + +/// The isolation invariant, stated as a test. +/// +/// `/memory` increments a counter held in the guest's linear memory and written +/// nowhere. A fresh instance has fresh memory, so it can only ever answer `1`. +/// Any other answer means two requests reached the same instance — and with +/// one instance behind two clients, the `Store` boundary that separates them +/// has moved from the runtime into guest code, where a single bug confuses two +/// clients instead of one. +/// +/// This is the test that fails if anyone reintroduces a shared instance. It +/// sends enough requests to mean it. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn no_two_requests_share_an_instance() { + let handle = handle_for(&[]).await; + for n in 1..=200 { + let (status, body) = get(&handle, "/memory").await; + assert_eq!(status, 200, "body: {body}"); + assert_eq!( + body.trim(), + "1", + "request {n} reached an instance a previous request had already used" + ); + } +} + +/// The guest is built before the listener binds, not on the first request. A +/// guest that will not instantiate should stop the enclave rather than serving +/// 500s while looking healthy. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn the_guest_is_proved_to_instantiate_at_startup() { + let handle = handle_for(&[]).await; + handle + .verify_instantiates() + .await + .expect("the guest instantiates"); + + // And keeps nothing: the check must not leave an instance behind for a + // request to inherit. + let (_, body) = get(&handle, "/memory").await; + assert_eq!( + body.trim(), + "1", + "the startup check left an instance behind" + ); +} + +/// A trap dies with its request. There is no instance to poison and nothing to +/// rebuild, so the next request is simply unaffected — which is the property +/// that a shared instance could not offer at any price. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_trap_dies_with_its_request() { + let handle = handle_for(&[]) + .await + .with_timeout(std::time::Duration::from_secs(3)); + + // `/hang` spins until the epoch traps it. + let req = hyper::Request::builder() + .method("GET") + .uri("http://enclave.test/hang") + .body(body(b"")) + .expect("well-formed request"); + let wedged = tokio::time::timeout( + std::time::Duration::from_secs(20), + handle.handle(Scheme::Http, req, None), + ) + .await + .expect("the watchdog did not fire"); + assert!(wedged.is_err(), "a wedged guest was reported as success"); + + let (status, body) = get(&handle, "/memory").await; + assert_eq!(status, 200, "the service did not recover: {body}"); + assert_eq!(body.trim(), "1"); +} + +// --------------------------------------------------------------------------- +// The client identity the guest is told about. +// --------------------------------------------------------------------------- + +/// A tenant id, as the gate would have resolved one. +/// +/// These tests exercise what happens *after* authentication — isolation, +/// warmth, concurrency — so they take the tenant as given. That the tenant can +/// only come from a verified assertion is the gate's own business, and +/// `auth::gate` tests it directly. +fn identity(name: &str) -> [u8; 16] { + let full = nitro_attestation::sha256(name.as_bytes()); + let mut id = [0u8; 16]; + id.copy_from_slice(&full[..16]); + id +} + +async fn whoami(handle: &ServeHandle, tenant: Option<&[u8; 16]>, forged: Option<&str>) -> String { + let mut builder = hyper::Request::builder() + .method("GET") + .uri("http://enclave.test/whoami"); + if let Some(value) = forged { + // Three copies, because one survivor is all a forgery needs and + // `append` would leave exactly that. + builder = builder + .header("x-enclave-client", value) + .header("x-enclave-client", value) + .header("x-enclave-client", value); + } + let req = builder.body(body(b"")).expect("well-formed request"); + let resp = handle + .handle(Scheme::Http, req, tenant) + .await + .expect("guest handled the request"); + let collected = resp.into_body().collect().await.expect("collecting body"); + String::from_utf8_lossy(&collected.to_bytes()) + .trim() + .to_string() +} + +/// The runtime's word reaches the guest. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn the_guest_is_told_which_tenant_is_calling() { + let handle = handle_for(&[]).await; + let id = identity("a-client"); + assert_eq!(whoami(&handle, Some(&id), None).await, hex::encode(id)); +} + +/// The forgery. A client sending the header itself must not be believed, and +/// sending it three times must not leave one behind — the guest's +/// `fields.get()` returns a list, so a survivor at `[0]` would be the +/// attacker's. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_client_cannot_forge_its_own_tenant() { + let handle = handle_for(&[]).await; + let real = identity("the-real-client"); + let stolen = hex::encode(identity("someone-else")); + + assert_eq!( + whoami(&handle, Some(&real), Some(&stolen)).await, + hex::encode(real), + "a client-supplied header was believed" + ); +} + +/// And the shorter route to the same forgery: no identity at all, so the +/// client's own header must be removed rather than passed through. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn an_unauthenticated_request_reaches_the_guest_as_anonymous() { + let handle = handle_for(&[]).await; + let stolen = hex::encode(identity("someone-else")); + + assert_eq!( + whoami(&handle, None, Some(&stolen)).await, + "(anonymous)", + "an unauthenticated client forged an identity" + ); +} + +// --------------------------------------------------------------------------- +// A filesystem per client. +// --------------------------------------------------------------------------- + +/// A handle that gives every authenticated client their own directory of the +/// one shared filesystem. +async fn tenanted() -> ServeHandle { + let handle = handle_for_with(&[]).await; + let tenancy = enclave_runtime::Tenancy::new(enclave_runtime::PoolLimits::default()); + handle.with_tenancy(Arc::new(tenancy)) +} + +/// A client's own directory persists between their requests, and the warm +/// instance means their in-memory state does too. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_client_keeps_their_own_directory_and_instance() { + let handle = tenanted().await; + let client = identity("a-client"); + + for expected in 1..=3 { + let (status, body) = get_as(&handle, "/counter", Some(&client)).await; + assert_eq!(status, 200, "body: {body}"); + assert_eq!(body.trim(), expected.to_string(), "the directory reset"); + } + // And the instance was kept, which a fresh one could never show. + for expected in 1..=3 { + let (_, body) = get_as(&handle, "/memory", Some(&client)).await; + assert_eq!( + body.trim(), + expected.to_string(), + "the instance was rebuilt" + ); + } +} + +/// A stream that is still moving bytes outlives the head timeout. +/// +/// The epoch deadline is a fixed budget of wall clock, and until the watchdog +/// learned to ask *why* a call was still running, that budget was the ceiling +/// on a stream's life: `/slow-stream` sends for ~2s and would be trapped part +/// way through by any timeout shorter than itself. Nothing about it is +/// unhealthy — the head went out at once and every chunk since has been read. +/// +/// The timeout here is deliberately far shorter than the stream, so passing +/// this cannot be an accident of timing. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_streaming_guest_outlives_the_request_timeout() { + let handle = handle_for_with(&[]) + .await + .with_timeout(std::time::Duration::from_millis(400)); + + let started = std::time::Instant::now(); + let (status, body) = get(&handle, "/slow-stream").await; + let took = started.elapsed(); + + assert_eq!(status, 200, "body: {body}"); + assert_eq!( + body, + (0..10).map(|n| format!("tick-{n}\n")).collect::(), + "the stream was cut short" + ); + assert!( + took > std::time::Duration::from_millis(400), + "the stream finished inside the timeout, so it proves nothing: {took:?}" + ); +} + +/// And the other half, which is what makes the first half safe. +/// +/// `/hang` spins in wasm without touching its body. It has no progress to show +/// and must still die on the old schedule — otherwise "a stream may run long" +/// would have become "anything may run forever", which is the trade the module +/// doc on `watchdog` refuses to make. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_spinning_guest_is_still_interrupted() { + let handle = handle_for_with(&[]) + .await + .with_timeout(std::time::Duration::from_secs(1)); + + let req = hyper::Request::builder() + .method("GET") + .uri("http://enclave.test/hang") + .body(body(&[])) + .expect("well-formed request"); + + let started = std::time::Instant::now(); + let result = handle.handle(Scheme::Http, req, None).await; + let took = started.elapsed(); + + assert!( + result.is_err(), + "a spinning guest was allowed to keep running" + ); + assert!( + took < std::time::Duration::from_secs(10), + "the spinning guest was not stopped promptly: {took:?}" + ); +} + +/// An abandoned call must not leave its instance in the tenant's slot. +/// +/// `/park` sleeps inside a host call, so there is no wasm executing for the +/// epoch to interrupt and `await_head` reaches its `abort`. Abort drops the +/// guest task *at its await*, which means nothing written after that await +/// runs — so the slot can only be left clean by code that already owns the +/// instance by then, not by a tidy-up that never executes. +/// +/// `/memory` is the probe: it counts in the guest's linear memory, so a reused +/// instance keeps counting and a rebuilt one starts over. Reading 2 here means +/// the next caller for this client got a store whose previous `call_handle` +/// was cancelled part-way through. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn an_abandoned_call_does_not_leave_its_instance_behind() { + let handle = tenanted() + .await + .with_timeout(std::time::Duration::from_millis(300)); + let client = identity("a-parked-client"); + + let (_, first) = get_as(&handle, "/memory", Some(&client)).await; + assert_eq!( + first.trim(), + "1", + "the first request should start the count" + ); + + // Abandoned, not trapped: the guest is parked where the epoch cannot see + // it, which is the whole point of this route. + let req = hyper::Request::builder() + .method("GET") + .uri("http://enclave.test/park") + .body(body(&[])) + .expect("well-formed request"); + let abandoned = handle.handle(Scheme::Http, req, Some(&client)).await; + assert!( + abandoned.is_err(), + "a guest parked past the timeout should have been abandoned" + ); + + let (status, after) = get_as(&handle, "/memory", Some(&client)).await; + assert_eq!(status, 200, "body: {after}"); + assert_eq!( + after.trim(), + "1", + "the abandoned call's instance was handed to the next request" + ); +} + +/// **The isolation property.** Two clients write the same path and neither +/// sees the other's value — because `/` means a different directory to each of +/// them, and the resolver will not let either name the other's. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn two_clients_never_see_each_others_data() { + let handle = tenanted().await; + let alice = identity("alice"); + let bob = identity("bob"); + + request_as( + &handle, + "POST", + "/files/secret.txt", + b"alice's", + Some(&alice), + ) + .await; + request_as(&handle, "POST", "/files/secret.txt", b"bob's", Some(&bob)).await; + + let (_, seen_by_alice) = get_as(&handle, "/files/secret.txt", Some(&alice)).await; + let (_, seen_by_bob) = get_as(&handle, "/files/secret.txt", Some(&bob)).await; + assert_eq!(seen_by_alice, "alice's"); + assert_eq!(seen_by_bob, "bob's", "one client read another's file"); + + // Their counters are independent too: same path, two directories. + let (_, a) = get_as(&handle, "/counter", Some(&alice)).await; + let (_, b) = get_as(&handle, "/counter", Some(&bob)).await; + assert_eq!(a.trim(), "1"); + assert_eq!(b.trim(), "1", "a shared counter means a shared view"); +} + +/// And no shared linear memory either — the `Store` boundary, not guest code. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn two_clients_never_share_an_instance() { + let handle = tenanted().await; + let alice = identity("alice"); + let bob = identity("bob"); + + for _ in 0..5 { + get_as(&handle, "/memory", Some(&alice)).await; + } + let (_, bobs_first) = get_as(&handle, "/memory", Some(&bob)).await; + assert_eq!( + bobs_first.trim(), + "1", + "a second client landed in the first client's instance" + ); +} + +/// An anonymous caller occupies no slot, so anyone who can open a socket +/// cannot fill the pool and evict every real client's mount. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn an_anonymous_caller_gets_no_tenant() { + let handle = tenanted().await; + for _ in 0..3 { + let (status, body) = get_as(&handle, "/memory", None).await; + assert_eq!(status, 200); + assert_eq!(body.trim(), "1", "an anonymous caller kept an instance"); + } +} + +/// A trap takes down one client's instance and touches nobody else — the whole +/// argument for the per-client boundary, as a test. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_trap_is_contained_to_one_client() { + let handle = tenanted() + .await + .with_timeout(std::time::Duration::from_secs(3)); + + let alice = identity("alice"); + let bob = identity("bob"); + + // Bob builds up instance state. + for _ in 0..3 { + get_as(&handle, "/memory", Some(&bob)).await; + } + + // Alice wedges hers. + let req = hyper::Request::builder() + .method("GET") + .uri("http://enclave.test/hang") + .body(body(b"")) + .expect("well-formed request"); + let _ = tokio::time::timeout( + std::time::Duration::from_secs(20), + handle.handle(Scheme::Http, req, Some(&alice)), + ) + .await + .expect("the watchdog did not fire"); + + // Alice's instance was rebuilt; Bob's was never touched. + let (status, alices) = get_as(&handle, "/memory", Some(&alice)).await; + assert_eq!(status, 200, "alice's tenant did not recover: {alices}"); + assert_eq!(alices.trim(), "1", "a trapped instance was reused"); + + let (_, bobs) = get_as(&handle, "/memory", Some(&bob)).await; + assert_eq!( + bobs.trim(), + "4", + "another client's trap reset bob's instance" + ); +} + +/// Their data survives too: a trap discards the poisoned instance and nothing +/// else. The filesystem is shared and was never in question. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_trap_keeps_the_clients_data() { + let handle = tenanted() + .await + .with_timeout(std::time::Duration::from_secs(3)); + let alice = identity("alice"); + + let (_, before) = get_as(&handle, "/counter", Some(&alice)).await; + assert_eq!(before.trim(), "1"); + + let req = hyper::Request::builder() + .method("GET") + .uri("http://enclave.test/hang") + .body(body(b"")) + .expect("well-formed request"); + let _ = tokio::time::timeout( + std::time::Duration::from_secs(20), + handle.handle(Scheme::Http, req, Some(&alice)), + ) + .await + .expect("the watchdog did not fire"); + + let (_, after) = get_as(&handle, "/counter", Some(&alice)).await; + assert_eq!( + after.trim(), + "2", + "the client's directory was lost after a trap" + ); +} + +/// The concurrency property, and the reason the whole design exists: two +/// clients run at the same time. `/hang` holds Alice's tenant until the epoch +/// traps it; Bob must be served long before that. +/// +/// Note what this does *not* do: abort the hung task and walk away. +/// `tokio::abort` needs an await point and spinning wasm never reaches one, so +/// only the epoch can stop it — and the epoch ticker is itself a task on this +/// runtime, which stops being polled once the test returns. Aborting and +/// leaving therefore hangs at shutdown instead of failing. The test waits for +/// the trap it knows is coming. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn one_client_does_not_wait_for_another() { + let handle = Arc::new( + tenanted() + .await + .with_timeout(std::time::Duration::from_secs(3)), + ); + let alice = identity("alice"); + let bob = identity("bob"); + + // Warm Bob first, so the assertion below measures contention rather than + // the cost of a cold mount. + assert_eq!(get_as(&handle, "/counter", Some(&bob)).await.0, 200); + + let hung = { + let handle = handle.clone(); + tokio::task::spawn(async move { + let req = hyper::Request::builder() + .method("GET") + .uri("http://enclave.test/hang") + .body(body(b"")) + .expect("well-formed request"); + let _ = handle.handle(Scheme::Http, req, Some(&alice)).await; + }) + }; + + tokio::time::sleep(std::time::Duration::from_millis(200)).await; + // Alice spins until the epoch traps her at ~2.5s. Queued behind her, Bob + // could not possibly answer inside 1.5s. + let served = tokio::time::timeout( + std::time::Duration::from_millis(1_500), + get_as(&handle, "/counter", Some(&bob)), + ) + .await + .expect("bob waited on alice's request; clients are not running concurrently"); + assert_eq!(served.0, 200); + assert_eq!(served.1.trim(), "2", "bob's own filesystem did not advance"); + + // Let the epoch trap Alice, so nothing is left spinning at shutdown. + tokio::time::timeout(std::time::Duration::from_secs(30), hung) + .await + .expect("the epoch never trapped the hung guest") + .expect("the hung task panicked"); +} + +/// **Separation is the runtime's, not the guest's.** +/// +/// `/escape/` opens whatever it is given with no validation at all — a +/// guest that is actively not helping. Everything below must still be refused, +/// and refused by the capability layer: absolute paths restart at the tenant's +/// own root, `..` stops there, and no number of either walks out. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_tenant_cannot_escape_its_directory_even_with_a_hostile_guest() { + let handle = tenanted().await; + let alice = identity("alice"); + let bob = identity("bob"); + + // Bob has something worth stealing, at a path Alice can name exactly. + request_as(&handle, "POST", "/files/secret.txt", b"bob's", Some(&bob)).await; + let bob_id = hex::encode(bob); + + for attempt in [ + format!("/tenants/{bob_id}/http-example/secret.txt"), + format!("../{bob_id}/http-example/secret.txt"), + format!("../../tenants/{bob_id}/http-example/secret.txt"), + "../../../../../../etc/passwd".to_string(), + ] { + let (status, body) = get_as(&handle, &format!("/escape/{attempt}"), Some(&alice)).await; + assert_eq!(status, 404, "{attempt} was not refused: {body}"); + assert!( + !body.contains("read 5 bytes"), + "{attempt} reached another tenant's file: {body}" + ); + } + + // And the same unvalidated route reads Alice's *own* file, so the refusals + // above are the scope working and not simply a broken code path. + request_as(&handle, "POST", "/files/mine.txt", b"alice's", Some(&alice)).await; + let (status, body) = get_as(&handle, "/escape//http-example/mine.txt", Some(&alice)).await; + assert_eq!(status, 200, "a tenant could not read its own file: {body}"); +} diff --git a/runtime/tests/serve_h2.rs b/runtime/tests/serve_h2.rs new file mode 100644 index 0000000..ebb0126 --- /dev/null +++ b/runtime/tests/serve_h2.rs @@ -0,0 +1,264 @@ +//! HTTP/2 on the wire, and HTTP/1.1 still on it beside. +//! +//! gRPC is HTTP/2 and nothing else, so a bidirectional stream to a guest needs +//! the listener to negotiate `h2`. What this suite pins is that it does, that +//! it does so without a configuration knob, and — the part that matters more — +//! that every client which never heard of h2 is untouched. The protocol is +//! chosen per connection from ALPN, so the two are not alternatives. +//! +//! ```console +//! $ (cd examples/guest-http && cargo build --release --target wasm32-wasip2) +//! $ cargo test -p enclave-runtime --test serve_h2 -- --include-ignored +//! ``` + +use std::path::PathBuf; +use std::sync::Arc; + +use enclave_runtime::{GuestEnvironment, HostClock, ServeConfig, TlsIdentity}; +use http_body_util::{BodyExt, Full}; +use s3fs_core::backend::memory::MemoryBackend; +use s3fs_core::{Config, Fs, MasterSecret}; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +fn component_path() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-http/target/wasm32-wasip2/release/guest-http.wasm") +} + +/// Every request carries one, whether or not the deployment attests. +fn nonce_header() -> String { + use base64::Engine as _; + use std::sync::atomic::{AtomicU64, Ordering}; + static NEXT: AtomicU64 = AtomicU64::new(1); + let mut nonce = vec![0xc3u8; 20]; + nonce[..8].copy_from_slice(&NEXT.fetch_add(1, Ordering::Relaxed).to_be_bytes()); + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(nonce) +} + +#[derive(Debug)] +struct AcceptAny; + +impl rustls::client::danger::ServerCertVerifier for AcceptAny { + fn verify_server_cert( + &self, + _end_entity: &rustls::pki_types::CertificateDer<'_>, + _intermediates: &[rustls::pki_types::CertificateDer<'_>], + _server_name: &rustls::pki_types::ServerName<'_>, + _ocsp: &[u8], + _now: rustls::pki_types::UnixTime, + ) -> Result { + Ok(rustls::client::danger::ServerCertVerified::assertion()) + } + fn verify_tls12_signature( + &self, + _message: &[u8], + _cert: &rustls::pki_types::CertificateDer<'_>, + _dss: &rustls::DigitallySignedStruct, + ) -> Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + fn verify_tls13_signature( + &self, + _message: &[u8], + _cert: &rustls::pki_types::CertificateDer<'_>, + _dss: &rustls::DigitallySignedStruct, + ) -> Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + fn supported_verify_schemes(&self) -> Vec { + rustls::crypto::aws_lc_rs::default_provider() + .signature_verification_algorithms + .supported_schemes() + } +} + +/// The runtime on an ephemeral port, TLS on, attestation off. +/// +/// Attestation is off deliberately: what this suite is about is the transport, +/// and a document on every response would only add a signature to each +/// assertion below without changing what any of them prove. +async fn start() -> std::net::SocketAddr { + let backend = Arc::new(MemoryBackend::new()); + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([5u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("creating the filesystem"); + + let bytes = std::fs::read(component_path()).unwrap_or_else(|e| { + panic!( + "reading {}: {e}\nbuild it first: (cd examples/guest-http && \ + cargo build --release --target wasm32-wasip2)", + component_path().display() + ) + }); + + let (logs, _collector) = + enclave_runtime::guest_io::start(Arc::new(enclave_runtime::TracingLogSink)); + let guest = GuestEnvironment::new( + fs, + Box::new(HostClock), + Arc::new(nitro_nsm::fake::FakeNsm::new()), + &[], + &[], + logs, + ) + .expect("guest environment"); + + let identity = + Arc::new(TlsIdentity::self_signed(&["enclave.test".to_string()]).expect("tls identity")); + + // Bind first so the test knows the port before the server owns it. + let listener = std::net::TcpListener::bind("127.0.0.1:0").expect("bind"); + let addr = listener.local_addr().unwrap(); + drop(listener); + + tokio::spawn(async move { + let _ = enclave_runtime::serve_component( + &bytes, + guest, + ServeConfig { + background_tasks: None, + notify: None, + addr, + certificate: Some(enclave_runtime::CertificateSlot::fixed(identity)), + acme: None, + attestation: None, + request_timeout: std::time::Duration::from_secs(30), + max_interaction: std::time::Duration::from_secs(300), + tenancy: None, + authentication: None, + egress: Default::default(), + }, + ) + .await; + }); + + let mut ready = false; + for _ in 0..400 { + if tokio::net::TcpStream::connect(addr).await.is_ok() { + ready = true; + break; + } + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + } + assert!(ready, "the server never came up on {addr}"); + addr +} + +/// A TLS connection offering exactly `alpn`, and what the server chose. +async fn connect( + addr: std::net::SocketAddr, + alpn: &[&[u8]], +) -> ( + tokio_rustls::client::TlsStream, + Option>, +) { + let mut config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .unwrap() + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAny)) + .with_no_client_auth(); + config.alpn_protocols = alpn.iter().map(|p| p.to_vec()).collect(); + + let connector = tokio_rustls::TlsConnector::from(Arc::new(config)); + let name = rustls::pki_types::ServerName::try_from("enclave.test").unwrap(); + let socket = tokio::net::TcpStream::connect(addr).await.expect("connect"); + let stream = connector.connect(name, socket).await.expect("handshake"); + let chosen = { + let (_, conn) = stream.get_ref(); + conn.alpn_protocol().map(|p| p.to_vec()) + }; + (stream, chosen) +} + +/// A client that asks for h2 is given h2. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn alpn_negotiates_h2_when_the_client_asks_for_it() { + let addr = start().await; + let (_stream, chosen) = connect(addr, &[b"h2"]).await; + assert_eq!( + chosen.as_deref(), + Some(&b"h2"[..]), + "a gRPC client offers only h2 and would have nothing to speak" + ); +} + +/// And a client that asks for HTTP/1.1 is still given HTTP/1.1. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn alpn_still_offers_http11() { + let addr = start().await; + let (_stream, chosen) = connect(addr, &[b"http/1.1"]).await; + assert_eq!(chosen.as_deref(), Some(&b"http/1.1"[..])); +} + +/// **The regression test for the whole transport change.** +/// +/// A client that never heard of ALPN sends no extension at all, and rustls +/// skips the negotiation entirely. Nothing about adding h2 may reach it: this +/// is a byte-for-byte HTTP/1.1 exchange of the kind every other suite makes. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_client_that_never_heard_of_alpn_is_unaffected() { + let addr = start().await; + let (mut stream, chosen) = connect(addr, &[]).await; + assert_eq!(chosen, None, "no ALPN offered, so none should be selected"); + + let request = format!( + "GET /counter HTTP/1.1\r\nHost: enclave.test\r\nx-enclave-nonce: {}\r\n\ + Connection: close\r\n\r\n", + nonce_header() + ); + stream.write_all(request.as_bytes()).await.expect("write"); + let mut response = Vec::new(); + let _ = stream.read_to_end(&mut response).await; + let text = String::from_utf8_lossy(&response); + assert!( + text.starts_with("HTTP/1.1 200"), + "an ordinary HTTP/1.1 client stopped working: {text}" + ); +} + +/// A request completes over h2, end to end, through the real guest. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_request_completes_over_h2() { + let addr = start().await; + let (stream, chosen) = connect(addr, &[b"h2"]).await; + assert_eq!(chosen.as_deref(), Some(&b"h2"[..])); + + let (mut sender, conn) = hyper::client::conn::http2::handshake( + hyper_util::rt::TokioExecutor::new(), + hyper_util::rt::TokioIo::new(stream), + ) + .await + .expect("h2 handshake"); + tokio::spawn(async move { + let _ = conn.await; + }); + + let req = hyper::Request::builder() + .method("GET") + .uri("https://enclave.test/counter") + .header("x-enclave-nonce", nonce_header()) + .body(Full::new(bytes::Bytes::new())) + .expect("well-formed request"); + + let resp = sender.send_request(req).await.expect("h2 request"); + assert_eq!(resp.status(), 200); + let body = resp.into_body().collect().await.expect("body").to_bytes(); + assert_eq!( + String::from_utf8_lossy(&body).trim(), + "1", + "the guest answered something else over h2" + ); +} diff --git a/runtime/tests/serve_tls.rs b/runtime/tests/serve_tls.rs new file mode 100644 index 0000000..98d490b --- /dev/null +++ b/runtime/tests/serve_tls.rs @@ -0,0 +1,719 @@ +//! The property this milestone exists for, end to end. +//! +//! A client opens a real TLS connection, asks for an attestation document +//! quoting a nonce it chose, verifies the document with the real verifier, and +//! checks that its `user_data` binds **the certificate it just saw in the +//! handshake**. If that holds, the connection terminates in the enclave that +//! signed the document — not in a proxy in front of it. +//! +//! The NSM here is a local fake, but it is not a stub: it produces a genuine +//! P-384 chain and a genuine ES384 COSE_Sign1 over a payload built from the +//! request it was given. The chain is pinned as the trust root and verified +//! against — `allow_untrusted_root` is off — so the verifier does the same +//! work it does in production. What the fake cannot supply is AWS's signing +//! key, so what these tests do *not* establish is that a real NSM's documents +//! chain to the real root. Only hardware shows that; the QEMU harness covers +//! the real device short of the AWS signature. +//! +//! ```console +//! $ (cd examples/guest-http && cargo build --release --target wasm32-wasip2) +//! $ cargo test -p enclave-runtime --test serve_tls -- --include-ignored +//! ``` + +use std::path::PathBuf; +use std::sync::{Arc, Mutex}; +use std::time::{Duration, SystemTime}; + +use enclave_runtime::{GuestEnvironment, HostClock, ServeConfig, TlsIdentity}; +use nitro_attestation::testing::TestChain; +use nitro_attestation::{AttestationHashes, Expectations, VerifyOptions}; +use nitro_nsm::{AttestationRequest, Nsm}; +use s3fs_core::backend::memory::MemoryBackend; +use s3fs_core::{Config, Fs, MasterSecret}; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +fn component_path() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("../examples/guest-http/target/wasm32-wasip2/release/guest-http.wasm") +} + +/// An NSM that signs for real, echoing back whatever it was asked to bind. +/// +/// A fake returning a canned document would let the endpoint pass while +/// binding the wrong certificate — precisely the bug these tests exist to +/// catch — so this builds the payload from the request. +#[derive(Debug)] +struct SigningNsm { + chain: TestChain, + pcr0: [u8; 48], + last: Mutex>, +} + +impl SigningNsm { + fn new() -> Self { + SigningNsm { + chain: TestChain::new().expect("test chain"), + pcr0: [0x5a; 48], + last: Mutex::new(None), + } + } +} + +impl Nsm for SigningNsm { + fn get_random(&self, buf: &mut [u8]) -> anyhow::Result<()> { + for (i, b) in buf.iter_mut().enumerate() { + *b = (i as u8).wrapping_mul(7).wrapping_add(3); + } + Ok(()) + } + + fn attest(&self, request: &AttestationRequest) -> anyhow::Result> { + *self.last.lock().unwrap() = Some(request.clone()); + self.chain + .document(request.user_data.clone(), request.nonce.clone(), self.pcr0) + } + + fn describe_pcr(&self, index: u16) -> anyhow::Result { + Ok(nitro_nsm::Pcr { + locked: index < 3, + value: if index == 0 { + self.pcr0.to_vec() + } else { + nitro_nsm::PCR_ZERO.to_vec() + }, + }) + } + + fn extend_pcr(&self, _index: u16, _data: &[u8]) -> anyhow::Result> { + anyhow::bail!("this fake does not model PCR extension") + } + + fn lock_pcr(&self, _index: u16) -> anyhow::Result<()> { + anyhow::bail!("this fake does not model PCR locking") + } + + fn describe(&self) -> String { + "signing test NSM".into() + } +} + +struct Harness { + addr: std::net::SocketAddr, + nsm: Arc, + certificate_der: Vec, + guest_bytes: Vec, +} + +/// Start the real [`enclave_runtime::serve_component`] on an ephemeral port. +/// Like [`start`], but serving from a slot the caller keeps — so a test can +/// replace the certificate while a connection is open. +async fn start_with_slot() -> (Harness, enclave_runtime::CertificateSlot) { + let slot = enclave_runtime::CertificateSlot::empty(); + let harness = start_inner(Some(slot.clone())).await; + (harness, slot) +} + +async fn start() -> Harness { + start_inner(None).await +} + +async fn start_inner(slot: Option) -> Harness { + let backend = Arc::new(MemoryBackend::new()); + // `create`, not `mount`: an empty store is a refusal since the boot + // machine landed, not an invitation to format one. + let fs = Fs::create( + backend.clone(), + backend, + &MasterSecret::from_bytes([9u8; 32]), + [0u8; 16], + Arc::new(Config::default()), + ) + .await + .expect("creating the filesystem"); + + let nsm = Arc::new(SigningNsm::new()); + let guest_bytes = std::fs::read(component_path()).unwrap_or_else(|e| { + panic!( + "reading {}: {e}\nbuild it first: (cd examples/guest-http && \ + cargo build --release --target wasm32-wasip2)", + component_path().display() + ) + }); + + // Detached deliberately: the collector runs for as long as this + // environment can send, which is what a test wants. Production drains it + // explicitly instead. + let (logs, _collector) = + enclave_runtime::guest_io::start(std::sync::Arc::new(enclave_runtime::TracingLogSink)); + let guest = GuestEnvironment::new(fs, Box::new(HostClock), nsm.clone(), &[], &[], logs) + .expect("guest environment"); + let identity = + Arc::new(TlsIdentity::self_signed(&["enclave.test".to_string()]).expect("tls identity")); + let certificate_der = identity.certificate_der.clone(); + // A caller that wants to replace the certificate later serves from its own + // slot; everyone else gets one that never changes. + let certificate = match &slot { + Some(slot) => { + slot.set(identity); + Some(slot.clone()) + } + None => Some(enclave_runtime::CertificateSlot::fixed(identity)), + }; + + // Bind first so the test knows the port before the server owns it. + let listener = std::net::TcpListener::bind("127.0.0.1:0").expect("bind"); + let addr = listener.local_addr().unwrap(); + drop(listener); + + let bytes = guest_bytes.clone(); + let nsm_for_server = nsm.clone(); + tokio::spawn(async move { + let _ = enclave_runtime::serve_component( + &bytes, + guest, + ServeConfig { + background_tasks: None, + notify: None, + addr, + certificate, + acme: None, + attestation: Some(nsm_for_server), + request_timeout: std::time::Duration::from_secs(30), + max_interaction: std::time::Duration::from_secs(300), + tenancy: None, + // These tests are about the attestation binding, which is + // reached under `/enclave/` and never passes the gate. + authentication: None, + egress: Default::default(), + }, + ) + .await; + }); + + // Compiling the guest takes a moment and several tests start servers at + // once. Wait properly, and say so on failure rather than leaving every + // assertion below to fail on a refused connection. + let mut ready = false; + for _ in 0..400 { + if tokio::net::TcpStream::connect(addr).await.is_ok() { + ready = true; + break; + } + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + } + assert!(ready, "the server never came up on {addr}"); + + Harness { + addr, + nsm, + certificate_der, + guest_bytes, + } +} + +/// A nonce, as a client is required to send one. +/// +/// Distinct per call rather than random: what these tests need is that no two +/// requests share one, so a document cannot pass for another request's. A real +/// client must use a CSPRNG. +fn fresh_nonce() -> Vec { + use std::sync::atomic::{AtomicU64, Ordering}; + static NEXT: AtomicU64 = AtomicU64::new(1); + let n = NEXT.fetch_add(1, Ordering::Relaxed); + let mut nonce = vec![0xa5u8; 20]; + nonce[..8].copy_from_slice(&n.to_be_bytes()); + nonce +} + +fn nonce_header(nonce: &[u8]) -> String { + use base64::Engine as _; + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(nonce) +} + +/// A TLS request, returning the status, body, and the certificate the server +/// presented — which is the whole point. Sends a fresh nonce, which every +/// request must now carry. +async fn https(addr: std::net::SocketAddr, path: &str) -> (u16, String, Vec) { + let (status, _, body, certificate) = https_full(addr, path, &fresh_nonce()).await; + (status, body, certificate) +} + +/// The same, plus the response head — which is where the per-response +/// attestation lives, and which `https` discards. +async fn https_full( + addr: std::net::SocketAddr, + path: &str, + nonce: &[u8], +) -> (u16, String, String, Vec) { + let config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .unwrap() + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAny)) + .with_no_client_auth(); + let connector = tokio_rustls::TlsConnector::from(Arc::new(config)); + let name = rustls::pki_types::ServerName::try_from("enclave.test").unwrap(); + + let socket = tokio::net::TcpStream::connect(addr).await.expect("connect"); + let mut stream = connector.connect(name, socket).await.expect("handshake"); + + let request = format!( + "GET {path} HTTP/1.1\r\nHost: enclave.test\r\nx-enclave-nonce: {}\r\n\ + Connection: close\r\n\r\n", + nonce_header(nonce) + ); + stream.write_all(request.as_bytes()).await.expect("write"); + + let mut response = Vec::new(); + let _ = stream.read_to_end(&mut response).await; + + let certificate = { + let (_, conn) = stream.get_ref(); + conn.peer_certificates() + .and_then(|c| c.first().cloned()) + .expect("server certificate") + .to_vec() + }; + + let split = response + .windows(4) + .position(|w| w == b"\r\n\r\n") + .expect("headers"); + let head = String::from_utf8_lossy(&response[..split]).to_string(); + let status: u16 = head + .lines() + .next() + .unwrap() + .split_whitespace() + .nth(1) + .unwrap() + .parse() + .unwrap(); + let raw = &response[split + 4..]; + // A guest response carries no content-length — the body streams out of the + // component — so hyper chunks it. + let body = if head + .to_ascii_lowercase() + .contains("transfer-encoding: chunked") + { + dechunk(raw) + } else { + raw.to_vec() + }; + ( + status, + head, + String::from_utf8_lossy(&body).to_string(), + certificate, + ) +} + +/// The attestation document from a response head, decoded. +fn document_from(head: &str) -> Option> { + use base64::Engine as _; + let line = head + .lines() + .find(|l| l.to_ascii_lowercase().starts_with("x-enclave-attestation:"))?; + let value = line.split_once(':')?.1.trim(); + base64::engine::general_purpose::STANDARD.decode(value).ok() +} + +fn dechunk(body: &[u8]) -> Vec { + let mut out = Vec::new(); + let mut rest = body; + while let Some(end) = rest.windows(2).position(|w| w == b"\r\n") { + let header = String::from_utf8_lossy(&rest[..end]); + let Ok(size) = usize::from_str_radix(header.split(';').next().unwrap_or("").trim(), 16) + else { + break; + }; + rest = &rest[end + 2..]; + if size == 0 || rest.len() < size { + break; + } + out.extend_from_slice(&rest[..size]); + rest = rest.get(size + 2..).unwrap_or(&[]); + } + out +} + +/// The guest keeps working over TLS, and never sees the runtime's paths. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn the_guest_serves_everything_else() { + let harness = start().await; + + let (status, body, _) = https(harness.addr, "/counter").await; + assert_eq!(status, 200, "{body}"); + assert_eq!(body.trim(), "1"); + + let (status, body, _) = https(harness.addr, "/counter").await; + assert_eq!(status, 200); + assert_eq!(body.trim(), "2", "state must persist across TLS requests"); +} + +#[derive(Debug)] +struct AcceptAny; + +impl rustls::client::danger::ServerCertVerifier for AcceptAny { + fn verify_server_cert( + &self, + _end_entity: &rustls::pki_types::CertificateDer<'_>, + _intermediates: &[rustls::pki_types::CertificateDer<'_>], + _server_name: &rustls::pki_types::ServerName<'_>, + _ocsp: &[u8], + _now: rustls::pki_types::UnixTime, + ) -> Result { + // Trust comes from the attestation binding, which these tests check + // explicitly; PKI has nothing to say about a certificate born inside + // an enclave ten milliseconds ago. + Ok(rustls::client::danger::ServerCertVerified::assertion()) + } + + fn verify_tls12_signature( + &self, + _message: &[u8], + _cert: &rustls::pki_types::CertificateDer<'_>, + _dss: &rustls::DigitallySignedStruct, + ) -> Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + + fn verify_tls13_signature( + &self, + _message: &[u8], + _cert: &rustls::pki_types::CertificateDer<'_>, + _dss: &rustls::DigitallySignedStruct, + ) -> Result { + Ok(rustls::client::danger::HandshakeSignatureValid::assertion()) + } + + fn supported_verify_schemes(&self) -> Vec { + rustls::crypto::aws_lc_rs::default_provider() + .signature_verification_algorithms + .supported_schemes() + } +} + +// --------------------------------------------------------------------------- +// Per-response attestation +// --------------------------------------------------------------------------- + +/// A request with no nonce is refused before anything runs — and the guest is +/// the witness: `/counter` increments only when the guest is invoked. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_request_without_a_nonce_never_reaches_the_guest() { + let harness = start().await; + + let (_, before, _) = https(harness.addr, "/counter").await; + + // The same request, minus the nonce. + let (status, head, body, _) = https_full(harness.addr, "/counter", b"").await; + assert_eq!(status, 400, "body: {body}"); + assert!(body.contains("x-enclave-nonce"), "body: {body}"); + assert!( + document_from(&head).is_none(), + "a refusal with no nonce has nothing to bind, so it must carry no document" + ); + + let (_, after, _) = https(harness.addr, "/counter").await; + assert_eq!( + after.trim().parse::().unwrap(), + before.trim().parse::().unwrap() + 1, + "the unnonced request reached the guest" + ); +} + +/// The guest cannot own the header. It sets a forged one on `/whoami`-style +/// routes and the client must still see exactly one, the runtime's. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn the_guest_cannot_forge_the_proof() { + let harness = start().await; + let nonce = fresh_nonce(); + let (status, head, _, _) = https_full(harness.addr, "/forge-attestation", &nonce).await; + assert_eq!(status, 200); + + // None, not one. The runtime does not attest a guest response — a caller + // has already identified the enclave on the `/auth/` exchange and pinned + // its certificate — so there is no document of its own to overwrite the + // guest's with. The header is taken away instead. + // + // This is the check that keeps the header runtime-owned. It used to hold + // as a side effect of attesting everything; now it is deliberate, and + // failing it means a guest can hand a client a document under the + // runtime's name. + let copies = head + .lines() + .filter(|l| l.to_ascii_lowercase().starts_with("x-enclave-attestation:")) + .count(); + assert_eq!(copies, 0, "the guest's copies survived:\n{head}"); +} + +/// A TLS connection the caller keeps, so it can ask twice. +/// +/// [`https_full`] sends `Connection: close` and reads to EOF, which is right +/// for a single request but makes "the same connection" impossible to express: +/// the socket is gone before the next line of the test runs. Renewal is only +/// interesting *while a connection is open*, so that case needs this. +struct KeptConnection { + stream: tokio_rustls::client::TlsStream, + /// The leaf from this connection's own handshake, captured once. + presented: Vec, +} + +impl KeptConnection { + async fn open(addr: std::net::SocketAddr) -> Self { + let config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .unwrap() + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAny)) + .with_no_client_auth(); + let connector = tokio_rustls::TlsConnector::from(Arc::new(config)); + let name = rustls::pki_types::ServerName::try_from("enclave.test").unwrap(); + let socket = tokio::net::TcpStream::connect(addr).await.expect("connect"); + let stream = connector.connect(name, socket).await.expect("handshake"); + let presented = { + let (_, conn) = stream.get_ref(); + conn.peer_certificates() + .and_then(|c| c.first().cloned()) + .expect("server certificate") + .to_vec() + }; + KeptConnection { stream, presented } + } + + /// One request, keep-alive, and the head of its response. + /// + /// Reads exactly one response rather than to EOF, so the connection is + /// still usable afterwards. + async fn request(&mut self, path: &str, nonce: &[u8]) -> String { + let request = format!( + "GET {path} HTTP/1.1\r\nHost: enclave.test\r\nx-enclave-nonce: {}\r\n\r\n", + nonce_header(nonce) + ); + self.stream + .write_all(request.as_bytes()) + .await + .expect("write"); + + let mut buf = Vec::new(); + let split = loop { + if let Some(at) = buf.windows(4).position(|w| w == b"\r\n\r\n") { + break at; + } + let mut chunk = [0u8; 4096]; + let n = self.stream.read(&mut chunk).await.expect("read"); + assert!(n > 0, "the connection closed before a response arrived"); + buf.extend_from_slice(&chunk[..n]); + }; + let head = String::from_utf8_lossy(&buf[..split]).to_string(); + + // Drain this response's body so the next request starts clean. Both + // framings appear here: runtime endpoints send a length, guest + // responses are chunked. + let lower = head.to_ascii_lowercase(); + let mut body = buf[split + 4..].to_vec(); + if lower.contains("transfer-encoding: chunked") { + while !body.ends_with(b"0\r\n\r\n") { + let mut chunk = [0u8; 4096]; + let n = self.stream.read(&mut chunk).await.expect("read body"); + assert!(n > 0, "the connection closed mid-body"); + body.extend_from_slice(&chunk[..n]); + } + } else { + let want = lower + .lines() + .find_map(|l| l.strip_prefix("content-length:")) + .and_then(|v| v.trim().parse::().ok()) + .unwrap_or(0); + while body.len() < want { + let mut chunk = [0u8; 4096]; + let n = self.stream.read(&mut chunk).await.expect("read body"); + assert!(n > 0, "the connection closed mid-body"); + body.extend_from_slice(&chunk[..n]); + } + } + head + } +} + +/// A connection keeps the certificate it was served, across a renewal. +/// +/// This is what the whole per-connection design is for, and the case a +/// globally-read certificate gets wrong: while connection A is open the slot is +/// replaced, so a runtime that read "the current certificate" at response time +/// would tell A about a certificate A was never served — and A's client, which +/// compares against its own handshake, would read that as an interceptor. +/// +/// The case that matters is A's **second** request, sent on the connection it +/// already had once the slot has moved on. A fresh connection after a renewal +/// proves nothing: it handshakes under the new certificate and is told about +/// the new certificate, which a runtime reading one global value gets right by +/// accident. Only a connection that outlives the replacement can catch it. +/// +/// Driven over `/auth/request/options` because that is where the runtime +/// attests now — the exchange a client uses to identify the enclave before +/// approving anything. This harness configures no gate, so the route itself +/// answers 404; the document on it is what the test is for, and it is attached +/// before the request is routed. +/// +/// `#[ignore]`d like its neighbours: [`start_with_slot`] builds the real +/// server, which needs the guest component built first. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs the guest component; see the module docs"] +async fn a_connection_keeps_the_certificate_it_was_served_across_a_renewal() { + let (harness, slot) = start_with_slot().await; + let original = harness.certificate_der.clone(); + + // A opens and STAYS OPEN for the rest of the test. + let mut a = KeptConnection::open(harness.addr).await; + assert_eq!(a.presented, original, "A was served the original"); + let nonce_a1 = fresh_nonce(); + let head_a1 = a.request("/auth/request/options", &nonce_a1).await; + + // The renewal, with A still connected. Everything that handshakes after + // this gets the new certificate; A must not. + let renewed = Arc::new( + TlsIdentity::self_signed(&["enclave.test".to_string()]).expect("a renewed identity"), + ); + assert_ne!(renewed.certificate_der, original); + slot.set(renewed.clone()); + + // The point of the whole test: A asks again, on the same connection, after + // the slot has moved on. + let nonce_a2 = fresh_nonce(); + let head_a2 = a.request("/auth/request/options", &nonce_a2).await; + + // B, served under the new certificate. + let nonce_b = fresh_nonce(); + let (_, head_b, _, presented_b) = + https_full(harness.addr, "/auth/request/options", &nonce_b).await; + assert_eq!( + presented_b, renewed.certificate_der, + "B should have been served the renewed certificate" + ); + + let options = VerifyOptions { + trust_root: harness.nsm.chain.root_der().to_vec(), + now: SystemTime::now(), + allow_untrusted_root: false, + }; + let check = |head: &str, nonce: &[u8], presented: &[u8]| { + let document = document_from(head).expect("a document"); + nitro_attestation::verify(&document, &options) + .expect("verifies") + .expect( + &Expectations { + nonce: Some(nonce.to_vec()), + user_data: Some( + AttestationHashes::new(presented, &harness.guest_bytes).serialize(), + ), + ..Default::default() + }, + SystemTime::now(), + ) + }; + + // Each document names the certificate its own connection was served. + check(&head_a1, &nonce_a1, &original).expect("A's document must bind A's certificate"); + check(&head_b, &nonce_b, &renewed.certificate_der) + .expect("B's document must bind the renewed certificate"); + + // The assertion the earlier shape could not make: A's *post-renewal* + // document still names the certificate A handshook with. A runtime reading + // "the current certificate" here would name the renewed one instead, and + // A's client — comparing against its own handshake — would read that as an + // interceptor sitting in front of the enclave. + check(&head_a2, &nonce_a2, &original) + .expect("A's second document must still bind the certificate A was served"); + assert!( + check(&head_a2, &nonce_a2, &renewed.certificate_der).is_err(), + "A was told about the renewed certificate it was never served" + ); + + // And the first document is not retroactively about the new certificate. + assert!( + check(&head_a1, &nonce_a1, &renewed.certificate_der).is_err(), + "A's document named a certificate A was never served" + ); +} + +/// The proof must not cost streaming. The document binds the nonce and the +/// connection's certificate — nothing the guest produces — so it is generated +/// before the guest is invoked, and the head goes out the moment the guest +/// sets it. +/// +/// `/trickle` sends three chunks with a pause between them. If the runtime +/// buffered the response, or waited for the body to finish before writing the +/// head, the head would not arrive until every chunk had. +#[tokio::test(flavor = "multi_thread")] +#[ignore = "needs examples/guest-http built for wasm32-wasip2"] +async fn a_document_does_not_wait_for_the_body() { + let harness = start().await; + let nonce = fresh_nonce(); + + let config = rustls::ClientConfig::builder_with_provider( + rustls::crypto::aws_lc_rs::default_provider().into(), + ) + .with_safe_default_protocol_versions() + .unwrap() + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAny)) + .with_no_client_auth(); + let connector = tokio_rustls::TlsConnector::from(Arc::new(config)); + let name = rustls::pki_types::ServerName::try_from("enclave.test").unwrap(); + let socket = tokio::net::TcpStream::connect(harness.addr) + .await + .expect("connect"); + let mut stream = connector.connect(name, socket).await.expect("handshake"); + + let request = format!( + "GET /trickle HTTP/1.1\r\nHost: enclave.test\r\nx-enclave-nonce: {}\r\n\ + Connection: close\r\n\r\n", + nonce_header(&nonce) + ); + let started = std::time::Instant::now(); + stream.write_all(request.as_bytes()).await.expect("write"); + + // Read only as far as the end of the head, so the timing below measures + // when the head arrived rather than when the whole response did. + let mut raw = Vec::new(); + let mut byte = [0u8; 1]; + while !raw.ends_with(b"\r\n\r\n") { + let n = stream.read(&mut byte).await.expect("read"); + assert!(n == 1, "the connection closed before the head was complete"); + raw.push(byte[0]); + } + let head_at = started.elapsed(); + let head = String::from_utf8_lossy(&raw).to_string(); + + let mut rest = Vec::new(); + let _ = stream.read_to_end(&mut rest).await; + let body_at = started.elapsed(); + let body = String::from_utf8_lossy(&dechunk(&rest)).to_string(); + + assert!( + head.to_ascii_lowercase() + .contains("transfer-encoding: chunked"), + "a streamed body must not be given a content-length: {head}" + ); + assert_eq!(body, "chunk-0\nchunk-1\nchunk-2\n", "{head}"); + + // The guest pauses 150ms before each of the last two chunks, so a + // buffering runtime could not have produced the head in under 300ms. + assert!( + body_at >= Duration::from_millis(250), + "the guest did not actually pause; the timing below proves nothing" + ); + assert!( + head_at < body_at / 2, + "the head waited for the body: head at {head_at:?}, body at {body_at:?}" + ); +} diff --git a/rust-toolchain.toml b/rust-toolchain.toml index 98d1403..e2dec5b 100644 --- a/rust-toolchain.toml +++ b/rust-toolchain.toml @@ -1,4 +1,8 @@ +# An exact release, not `stable`. CI denies every clippy warning, and each +# stable release adds lints: a floating channel turned a green branch red with +# no change to it. Moving to a newer release is a commit of its own, made with +# whatever its clippy asks for. [toolchain] -channel = "stable" +channel = "1.98.0" components = ["rustfmt", "clippy"] targets = ["wasm32-wasip2"] diff --git a/scripts/build-guest.sh b/scripts/build-guest.sh new file mode 100755 index 0000000..f440bc9 --- /dev/null +++ b/scripts/build-guest.sh @@ -0,0 +1,32 @@ +#!/usr/bin/env bash +# Build one example guest to a wasm32-wasip2 component. +# +# build-guest.sh http | grpc | sqlite +# +# Each example is its own cargo workspace, so these do not share the root +# target/ directory and have to be built by name rather than swept up by +# `--workspace`. + +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +guest="${1:?which guest: http, grpc or sqlite}" +dir="$REPO/examples/guest-$guest" +[[ -d "$dir" ]] || fail "no such guest: examples/guest-$guest" + +cd "$dir" + +# Only the SQLite guest compiles C. Its sysroot comes from wasi-sdk, which +# scripts/wasi-sdk.sh puts in place; the rest of the guests need no C toolchain +# at all, which is why this is a special case rather than the default. +if [[ "$guest" == "sqlite" ]]; then + sdk="${WASI_SDK:-$HOME/wasi-sdk}" + [[ -x "$sdk/bin/clang" ]] || fail "no wasi-sdk at $sdk — run scripts/wasi-sdk.sh first" + export CC_wasm32_wasip2="$sdk/bin/clang" + export AR_wasm32_wasip2="$sdk/bin/ar" + export CFLAGS_wasm32_wasip2="--sysroot=$sdk/share/wasi-sysroot -DSQLITE_THREADSAFE=0 -DHAVE_USLEEP=1" +fi + +say "building guest-$guest" +cargo build --release --target wasm32-wasip2 + +echo "built $dir/target/wasm32-wasip2/release/guest-$guest.wasm" diff --git a/scripts/ci-bench.sh b/scripts/ci-bench.sh new file mode 100755 index 0000000..e79e6c4 --- /dev/null +++ b/scripts/ci-bench.sh @@ -0,0 +1,38 @@ +#!/usr/bin/env bash +# The two criterion benches, in the form the tracker can read. +# +# `instance_cost` measures what a warm instance saves over a fresh one, and it +# runs the real component — so the guest has to be built before cargo is asked +# for a number, or the bench panics with a build hint instead of producing one. +# `fs_hot_paths` runs over the in-memory backend and needs nothing. +# +# `--output-format bencher` is criterion's libtest-compatible output, which is +# what github-action-benchmark reads as `tool: cargo`. Criterion's own format is +# richer and completely opaque to it. +# +# Worth saying plainly: numbers from a shared runner carry its neighbours in +# them. The tracked history is for spotting a change of shape over many runs, +# not for trusting any single figure — which is why the alert threshold in the +# workflow is deliberately loose and never fails the build. + +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" +cd "$REPO" + +"$REPO/scripts/build-guest.sh" http + +say "benchmarks" +# Each bench target is named rather than swept up with `--workspace`. Everything +# after `--` is handed to *every* target cargo runs, and in bench profile that +# includes each crate's libtest harness — which does not know `--output-format`, +# rejects it, and fails the whole run before a single benchmark executes. Naming +# the two `harness = false` criterion targets keeps the flag with the only +# harnesses that understand it. +# +# lib.sh sets `pipefail`, so a failing cargo still fails the script despite tee. +: > "$REPO/bench.txt" +cargo bench -p enclave-runtime --bench instance_cost -- --output-format bencher \ + | tee -a "$REPO/bench.txt" +cargo bench -p s3fs-core --bench fs_hot_paths -- --output-format bencher \ + | tee -a "$REPO/bench.txt" + +echo "wrote $REPO/bench.txt" diff --git a/scripts/ci-check.sh b/scripts/ci-check.sh new file mode 100755 index 0000000..8289d29 --- /dev/null +++ b/scripts/ci-check.sh @@ -0,0 +1,25 @@ +#!/usr/bin/env bash +# Formatting, lints, and every test that needs nothing built beside it. +# +# The fast gate. Nothing here needs a wasm guest, Docker, or a network, so it is +# the job that should fail first when something is simply wrong. + +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" +cd "$REPO" + +say "rustfmt" +cargo fmt --all -- --check + +# `--all-targets` so the lints cover tests and benches too, and `-D warnings` +# because a lint nobody has to fix is a lint that accumulates. +say "clippy" +cargo clippy --workspace --all-targets -- -D warnings + +say "unit tests" +cargo test --workspace --lib + +# Integration suites run only where they are named, and this one ran nowhere +# before. It is the boot machine — genesis, resume, upgrades and the refusals — +# over the in-memory backend, so it needs no guest built. +say "boot machine" +cargo test -p enclave-runtime --test boot_origin diff --git a/scripts/ci-e2e.sh b/scripts/ci-e2e.sh new file mode 100755 index 0000000..b853aaa --- /dev/null +++ b/scripts/ci-e2e.sh @@ -0,0 +1,67 @@ +#!/usr/bin/env bash +# The whole stack, in an emulated enclave. +# +# deploy/qemu-nitro/run-e2e.sh is the harness and asserts its own preconditions; +# this script is what puts them in place on a machine that has none of them. On +# a workstation that already boots enclaves, everything below is a no-op and the +# harness runs directly. +# +# Nested virtualisation on GitHub-hosted runners works but is not supported by +# GitHub, so this is the one job here that can fail for reasons outside this +# repository. Everything it needs is installed explicitly rather than assumed, +# so that when it does fail the log says which part was missing. + +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" +cd "$REPO" + +WORK="${WORK:-$REPO/target/qemu-nitro}" +IMAGE="${QEMU_IMAGE:-s3fs-qemu-nitro:latest}" + +# Pinned, like the Pebble image beside it. An unpinned build tool makes the +# harness's behaviour depend on whatever crates.io served that morning, which is +# the one thing here that was still floating. +VSOCK_VERSION="${VSOCK_VERSION:-0.3.0}" + +# The in-process suites first, against the same guest, before anything is +# booted. They are minutes where the emulated enclave is the better part of an +# hour, and they fail on the layer at fault — a TLS or gRPC framing bug found +# here names itself, where the same bug found through QEMU is a timeout with a +# console log to read. Folded into this job rather than kept as their own so +# there is one integration signal, not two. +"$REPO/scripts/ci-guests.sh" + +# The block store against a real S3 implementation, folded in for the same +# reason. It starts its own MinIO through testcontainers, so it needs nothing +# from the harness below and nothing from the store the enclave will mount. +"$REPO/scripts/ci-storage.sh" + +# /dev/kvm exists on hosted runners but is root-owned; the runner user needs it. +# Guarded so a workstation where KVM already works is left alone — this rule is +# a CI accommodation, not something to apply to a developer's machine. +if [[ -e /dev/kvm && ! -r /dev/kvm ]]; then + say "granting access to /dev/kvm" + echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' \ + | sudo tee /etc/udev/rules.d/99-kvm4all.rules >/dev/null + sudo udevadm control --reload-rules + sudo udevadm trigger --name-match=kvm +fi + +# The harness forwards the enclave's vsock connections to host loopback, which +# needs the loopback transport present. +if [[ ! -e /dev/vsock ]]; then + say "loading vsock_loopback" + sudo modprobe vsock_loopback || fail "could not load vsock_loopback" +fi + +if ! docker image inspect "$IMAGE" >/dev/null 2>&1; then + say "building the QEMU image" + docker build -t "$IMAGE" deploy/qemu-nitro +fi + +# Idempotent, and cached in CI by the directory it installs into. +if [[ ! -x "$WORK/tools/bin/vhost-device-vsock" ]]; then + say "installing vhost-device-vsock $VSOCK_VERSION" + cargo install vhost-device-vsock --version "$VSOCK_VERSION" --root "$WORK/tools" --locked +fi + +exec deploy/qemu-nitro/run-e2e.sh diff --git a/scripts/ci-guests.sh b/scripts/ci-guests.sh new file mode 100755 index 0000000..4dee6f4 --- /dev/null +++ b/scripts/ci-guests.sh @@ -0,0 +1,77 @@ +#!/usr/bin/env bash +# Serving a real guest: dispatch, transport, the gate, and background work. +# +# Everything here runs the real component through the real linker over the +# in-memory backend, so the whole dispatch path is covered without Docker or a +# network. The suites are `#[ignore]`d because they need a guest built first, +# which is why every one of them is run with `--include-ignored`. +# +# Every suite is named explicitly. There is no glob, so a suite nobody lists +# runs nowhere — which is how `serve_auth`, the suite for the entire +# authorization model, went uncovered for as long as it did. + +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" +cd "$REPO" + +say "WIT drift" +"$REPO/scripts/wit-drift.sh" + +"$REPO/scripts/build-guest.sh" http +"$REPO/scripts/build-guest.sh" grpc + +# Built before the suites rather than between them: one of serve_auth's tests +# drives this real binary as a subprocess instead of a stand-in, and skips +# silently when it is missing. A skipped test that reports green is the failure +# mode worth spending a build on avoiding. +say "building the passkey client" +cargo build --release -p enclave-runtime --features testing --bin passkey-client + +# A binary's own unit tests are not part of `--lib`, so they need naming. These +# are the measurements the client insists on before it approves anything. +# `--release` so this shares the compile with the binary just built. +say "passkey client" +cargo test --release -p enclave-runtime --features testing --bin passkey-client + +say "background tasks" +cargo test -p enclave-runtime --lib tasks::tests -- --include-ignored + +say "held connections" +cargo test -p enclave-runtime --lib stream::tests -- --include-ignored + +say "serving a guest" +cargo test -p enclave-runtime --test serve_guest -- --include-ignored + +# The central claim: a client opens a real TLS connection and checks that the +# attestation document binds the certificate it just saw. +say "TLS termination and the attestation binding" +cargo test -p enclave-runtime --test serve_tls -- --include-ignored + +# h2 is negotiated for the clients that ask, and every client that does not is +# untouched. +say "HTTP/2 negotiation, and HTTP/1.1 beside it" +cargo test -p enclave-runtime --test serve_h2 -- --include-ignored + +# Both directions open at once: the guest answering while the client is still +# sending, backpressure, trailers, and what an open stream costs the tenant +# holding it. +say "bidirectional gRPC streaming" +cargo test -p enclave-runtime --test serve_grpc -- --include-ignored + +# The same guest driven by tonic over TLS, so the guest's hand-written framing +# is checked against an implementation from outside this repo. +say "gRPC on the wire, with a real client" +cargo test -p enclave-runtime --test serve_grpc_wire -- --include-ignored + +# The whole authorization model — challenge, assertion, token, scope, tenant +# isolation — and the only suite that checks an attestation document against the +# connection it arrived on. +say "interaction tokens, end to end" +cargo test -p enclave-runtime --test serve_auth -- --include-ignored + +say "guest logging" +cargo test -p enclave-runtime --test guest_logs -- --include-ignored + +# The verifier's own tests build the binary, so there is no separate build step +# here — nothing in CI executes the artifact itself. +say "verifier measurement requirements" +cargo test -p nitro-attestation --features cli --bin nitro-attest diff --git a/scripts/ci-sqlite.sh b/scripts/ci-sqlite.sh new file mode 100755 index 0000000..6c46767 --- /dev/null +++ b/scripts/ci-sqlite.sh @@ -0,0 +1,80 @@ +#!/usr/bin/env bash +# A real C database driving the filesystem, and the benchmark table it prints. +# +# Page-granular random I/O, a rollback journal created and unlinked per +# transaction, VACUUM rewriting the whole file, and PRAGMA integrity_check to +# say whether the bytes came back correct. Nothing else in CI exercises the +# block store this way. +# +# The runtime serves `wasi:http/proxy` and nothing else, so the workload is +# asked for with a request rather than run as a process. TLS off and no relying +# party configured: this is about SQLite over the block store, not about the +# gate, which ci-guests.sh covers. +# +# **Not wired into CI, and it cannot be.** Since `bbbee4d` the runtime measures +# its guest into PCR16 before serving, and that reads the NSM: +# +# runtime failed to start the guest +# error="reading PCR16 before measuring the guest: no PCRs: entropy is +# coming from the kernel, not an NSM" +# +# `measure_guest` is called unconditionally and has no flag to skip it — which +# is the point of it. A hosted runner has no /dev/nsm, so this script only runs +# where an enclave does. Run it by hand there; do not add it back to the +# workflow expecting it to pass. + +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" +cd "$REPO" + +MINIO_CONTAINER=ci-minio-sql +MINIO_PORT=9002 + +"$REPO/scripts/wasi-sdk.sh" +"$REPO/scripts/build-guest.sh" sqlite + +say "building enclave-runtime" +cargo build --release -p enclave-runtime + +"$REPO/scripts/minio-up.sh" "$MINIO_CONTAINER" "$MINIO_PORT" sql-data sql-roots +trap '"$REPO/scripts/minio-down.sh" "$MINIO_CONTAINER"' EXIT + +say "running the SQLite workload" +export SQLITE_SCALE="${SQLITE_SCALE:-20000}" +export S3FS_MASTER_KEY="00000000000000000000000000000000000000000000000000000000000000ef" +# Required, with no default: the runtime will not guess where a master key comes +# from. `static` is the development source that reads the key above — the same +# one the QEMU emulator image uses, and for the same reason, since KMS will not +# release a key against an unsigned attestation document. +export S3FS_MASTER_KEY_SOURCE=static +export S3FS_BUCKET=sql-data +export S3FS_ROOTS_BUCKET=sql-roots +export S3FS_ENDPOINT="http://127.0.0.1:$MINIO_PORT" +export S3FS_FORCE_PATH_STYLE=1 +export AWS_ACCESS_KEY_ID=minioadmin +export AWS_SECRET_ACCESS_KEY=minioadmin +export AWS_REGION=us-east-1 +export S3FS_GUEST_PATH=examples/guest-sqlite/target/wasm32-wasip2/release/guest-sqlite.wasm +export S3FS_TLS=off +export S3FS_HTTP_LISTEN=127.0.0.1:8080 + +./target/release/enclave-runtime > runtime.log 2>&1 & +runtime=$! +trap 'kill "$runtime" 2>/dev/null || true; "$REPO/scripts/minio-down.sh" "$MINIO_CONTAINER"' EXIT + +for _ in $(seq 1 60); do + curl -sf -o /dev/null http://127.0.0.1:8080/ && break + kill -0 "$runtime" 2>/dev/null || { echo "the runtime exited:"; cat runtime.log; exit 1; } + sleep 1 +done + +# No --max-time: the workload is a long request by design, and the runtime's own +# --request-timeout is what bounds it. A wedged listener is bounded by the CI +# job's timeout-minutes instead, which covers the cases a client timeout cannot. +out=$(curl -sf http://127.0.0.1:8080/) +echo "$out" +echo "$out" | grep -qx OK + +# The timing table went to the guest's stdout, which the runtime frames into its +# own log records. +echo "--- runtime log ---" +cat runtime.log diff --git a/scripts/ci-storage.sh b/scripts/ci-storage.sh new file mode 100755 index 0000000..6344cbb --- /dev/null +++ b/scripts/ci-storage.sh @@ -0,0 +1,14 @@ +#!/usr/bin/env bash +# The block store against a real S3 implementation. +# +# These tests start their own MinIO through testcontainers — one per test, which +# is why they run single-threaded and take a while. That is also why this script +# does not call minio-up.sh: a store started here would sit unused. + +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" +cd "$REPO" + +"$REPO/scripts/minio-image.sh" >/dev/null + +say "MinIO integration" +cargo test -p s3fs-core --features aws --test minio_integration -- --ignored --test-threads=1 diff --git a/scripts/lib.sh b/scripts/lib.sh new file mode 100755 index 0000000..cb0bcc7 --- /dev/null +++ b/scripts/lib.sh @@ -0,0 +1,25 @@ +#!/usr/bin/env bash +# Common to every script here. +# +# These recipes used to live inline in .github/workflows/ci.yml, which meant the +# only way to run a CI job was to push and wait. Everything in this directory +# runs the same way on a laptop as it does on a runner — that is the whole point +# of the directory existing. +# +# Sourced, not executed: `. "$(dirname "$0")/lib.sh"`. + +set -euo pipefail + +# Not `dirname $0/..`: a script invoked through a symlink or from another +# directory would resolve somewhere else, and every path below is repo-relative. +REPO="$(git rev-parse --show-toplevel)" +export REPO + +# The same shape deploy/qemu-nitro/run-e2e.sh prints, so a run that spans both +# reads as one log rather than two conventions meeting in the middle. +say() { printf '\n== %s ==\n' "$*"; } + +fail() { + printf '\nFAIL: %s\n' "$*" >&2 + exit 1 +} diff --git a/scripts/minio-down.sh b/scripts/minio-down.sh new file mode 100755 index 0000000..9c756df --- /dev/null +++ b/scripts/minio-down.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +# Stop a MinIO started by minio-up.sh. +# +# minio-down.sh +# +# Never fails. It runs from a trap and from an `if: always()` step, where an +# error of its own would mask the failure that is actually worth reading. + +set -euo pipefail + +docker rm -f "${1:?container name}" >/dev/null 2>&1 || true diff --git a/scripts/minio-image.sh b/scripts/minio-image.sh new file mode 100755 index 0000000..6af192a --- /dev/null +++ b/scripts/minio-image.sh @@ -0,0 +1,50 @@ +#!/usr/bin/env bash +# The MinIO image every store here runs on, built from source. +# +# image="$(scripts/minio-image.sh)" +# +# MinIO no longer publishes images anyone can pull: Docker Hub refuses +# `minio/minio` without a login, and quay.io has no tags at all. The releases +# are still tagged on GitHub, so the pinned one is built here — the server, and +# `mc` for the scripts that make buckets and upload guests — under a name nobody +# will mistake for an upstream image. The tag is the release the testcontainers +# module pins, so the integration tests only have to change the name. +# +# Builds only when the image is missing. Prints the image on stdout and +# everything else on stderr, so a caller can take the name and nothing more. + +set -euo pipefail + +MINIO_RELEASE=RELEASE.2025-02-28T09-55-16Z +# The last mc release before that server. +MC_RELEASE=RELEASE.2025-02-21T16-00-46Z +IMAGE="enclave-runtime/minio:$MINIO_RELEASE" + +if ! docker image inspect "$IMAGE" >/dev/null 2>&1; then + # ponytail: rebuilt on every CI run, a few minutes; `docker save` it into + # an actions/cache keyed on this script if those minutes start to matter. + printf '\n== building %s from source ==\n' "$IMAGE" >&2 + docker build -t "$IMAGE" \ + --build-arg MINIO_RELEASE="$MINIO_RELEASE" \ + --build-arg MC_RELEASE="$MC_RELEASE" - >&2 <<'EOF' +FROM golang:1.23-alpine AS build +RUN apk add --no-cache git && mkdir /out +ARG MINIO_RELEASE +ARG MC_RELEASE +# Each project's own version stamping, so a log says which release is running. +RUN git clone --quiet --depth 1 --branch "$MINIO_RELEASE" https://github.com/minio/minio /src/minio \ + && cd /src/minio \ + && CGO_ENABLED=0 go build -tags kqueue -trimpath -ldflags "$(go run buildscripts/gen-ldflags.go)" -o /out/minio . +RUN git clone --quiet --depth 1 --branch "$MC_RELEASE" https://github.com/minio/mc /src/mc \ + && cd /src/mc \ + && CGO_ENABLED=0 go build -trimpath -ldflags "$(go run buildscripts/gen-ldflags.go)" -o /out/mc . + +FROM alpine:3.20 +COPY --from=build /out/ /usr/bin/ +# Declared, as the upstream image declares it: testcontainers publishes only +# the ports an image exposes. +EXPOSE 9000 +ENTRYPOINT ["minio"] +EOF +fi +echo "$IMAGE" diff --git a/scripts/minio-up.sh b/scripts/minio-up.sh new file mode 100755 index 0000000..5b15f26 --- /dev/null +++ b/scripts/minio-up.sh @@ -0,0 +1,75 @@ +#!/usr/bin/env bash +# Start MinIO and create the two buckets the runtime expects. +# +# minio-up.sh [data-dir] +# +# Without data-dir the store lives inside the container and goes with it, which +# is what tests want. With one it lives in that host directory and outlives the +# container, so a later start with the same directory serves the same store. +# +# The SQLite CI job and deploy/qemu-nitro/run-e2e.sh each carried their own copy +# of this. Two recipes that have to agree, with nothing making them agree, is +# how the harness and CI end up testing subtly different stores. +# +# The roots bucket is created `--with-lock` deliberately: object lock is what +# makes a published root record impossible to roll back, so a store without it +# would pass tests that a real deployment could not. + +set -euo pipefail + +container="${1:?container name}" +port="${2:?host port}" +data="${3:?data bucket}" +roots="${4:?roots bucket}" +data_dir="${5:-}" + +# Built from source the first time; minio-image.sh says why. +image="$("$(dirname "${BASH_SOURCE[0]}")/minio-image.sh")" + +# Optional, and applied to every container this starts. A caller that cleans up +# by label — deploy/qemu-nitro/lib.sh does, so that the list of containers to +# remove cannot fall out of step with the list it starts — otherwise leaves this +# one running, because it is the one container it did not start itself. +label=() +[[ -n "${MINIO_LABEL:-}" ]] && label=(--label "$MINIO_LABEL") +volume=() +if [[ -n "$data_dir" ]]; then + mkdir -p "$data_dir" + # As the invoking user, so the directory can be deleted without root. + volume=(-v "$data_dir:/data" --user "$(id -u):$(id -g)") +fi + +docker rm -f "$container" >/dev/null 2>&1 || true +# MINIO_BIND publishes the port on one address only — 127.0.0.1 on a host that is reachable from +# the internet, where the store's well-known credentials would otherwise be everyone's. The enclave +# still reaches it: gvproxy delivers its host address, 192.168.127.254, to the host's loopback. +docker run -d --rm --name "$container" -p "${MINIO_BIND:+$MINIO_BIND:}$port:9000" "${label[@]}" "${volume[@]}" \ + -e MINIO_ROOT_USER=minioadmin -e MINIO_ROOT_PASSWORD=minioadmin \ + "$image" server /data >/dev/null + +# Ready, not merely started. `mc` against a half-open MinIO fails in ways that +# read as a bucket problem rather than a timing one, which is a bad half hour +# for whoever reads the log next. +ready="" +for _ in $(seq 60); do + if curl -sf "http://127.0.0.1:$port/minio/health/ready" >/dev/null; then + ready=1 + break + fi + sleep 1 +done +if [[ -z "$ready" ]]; then + echo "MinIO never became ready on :$port" >&2 + docker logs "$container" 2>&1 | tail -20 >&2 + exit 1 +fi + +# Tolerant of buckets that already exist: this script is also how a developer +# restarts a store between runs, and "already there" is success. +docker run --rm --network host "${label[@]}" --entrypoint sh "$image" -c " + mc alias set m http://127.0.0.1:$port minioadmin minioadmin >/dev/null + mc mb m/$data >/dev/null 2>&1 || true + mc mb --with-lock m/$roots >/dev/null 2>&1 || true" >/dev/null \ + || { echo "could not create $data and $roots" >&2; exit 1; } + +echo "MinIO ready on :$port with $data and $roots" diff --git a/scripts/wasi-sdk.sh b/scripts/wasi-sdk.sh new file mode 100755 index 0000000..aad1933 --- /dev/null +++ b/scripts/wasi-sdk.sh @@ -0,0 +1,49 @@ +#!/usr/bin/env bash +# Put wasi-sdk where the SQLite guest's C toolchain can find it. +# +# Idempotent: with the SDK already unpacked this exits immediately, which is +# what makes the CI cache worth having and what lets a developer run +# ci-sqlite.sh repeatedly without re-downloading 110 MB. +# +# The download used to be `curl -sSL` with no `-f`. Without it a non-2xx +# response is written to the tarball as HTML and the failure surfaces two steps +# later inside `tar`, complaining about the archive format and naming nothing +# that would lead you to the real cause. + +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +WASI_SDK_VERSION="${WASI_SDK_VERSION:-25}" +WASI_SDK_RELEASE="${WASI_SDK_RELEASE:-25.0}" +WASI_SDK="${WASI_SDK:-$HOME/wasi-sdk}" + +# wasi-sdk publishes no checksum file with its releases — the assets are the +# tarballs and nothing else — so this is the hash of the artifact this repo was +# tested against, recorded here rather than taken on trust from the network +# every run. Recompute with: +# sha256sum wasi-sdk-25.0-x86_64-linux.tar.gz +WASI_SDK_SHA256="${WASI_SDK_SHA256:-52640dde13599bf127a95499e61d6d640256119456d1af8897ab6725bcf3d89c}" + +if [[ -x "$WASI_SDK/bin/clang" ]]; then + echo "wasi-sdk already at $WASI_SDK" + exit 0 +fi + +say "installing wasi-sdk $WASI_SDK_RELEASE" + +tarball="$(mktemp -d)/wasi-sdk.tar.gz" +url="https://github.com/WebAssembly/wasi-sdk/releases/download/wasi-sdk-${WASI_SDK_VERSION}/wasi-sdk-${WASI_SDK_RELEASE}-x86_64-linux.tar.gz" + +# --retry-all-errors so a transient 5xx from the release CDN is a pause rather +# than a failed run; -f so an error page is an error, not a tarball. +curl -fL --retry 3 --retry-all-errors --max-time 600 -o "$tarball" "$url" \ + || fail "could not download $url" + +echo "$WASI_SDK_SHA256 $tarball" | sha256sum -c - \ + || fail "wasi-sdk checksum mismatch — the release was changed, or the download was truncated" + +mkdir -p "$WASI_SDK" +tar xf "$tarball" -C "$WASI_SDK" --strip-components=1 +rm -rf "$(dirname "$tarball")" + +[[ -x "$WASI_SDK/bin/clang" ]] || fail "unpacked wasi-sdk has no bin/clang at $WASI_SDK" +echo "wasi-sdk $WASI_SDK_RELEASE installed at $WASI_SDK" diff --git a/scripts/wit-drift.sh b/scripts/wit-drift.sh new file mode 100755 index 0000000..143ab3b --- /dev/null +++ b/scripts/wit-drift.sh @@ -0,0 +1,63 @@ +#!/usr/bin/env bash +# Every vendored WIT copy must equal the canonical one under wit/. +# +# A guest vendors its own copy because wit-bindgen reads a path inside the +# guest's crate. Two copies of an interface that must be byte-identical is +# exactly the thing that drifts unnoticed: both sides still compile, and the +# mismatch surfaces much later as a runtime trap in a component that looked fine. +# +# Only `wit/deps/` is checked, which is not a shortcut but the rule WIT itself +# imposes: every file directly in a guest's `wit/` belongs to that guest's own +# package, and foreign packages must live under `deps/`. So `deps/` is exactly +# the vendored set, and a guest's own world file is correctly ignored. +# +# A loop rather than the single hard-coded `cmp` this replaces. The next guest +# to vendor a copy would otherwise have been silently uncovered. + +set -euo pipefail + +REPO="$(git rev-parse --show-toplevel)" +cd "$REPO" + +status=0 +found=0 + +while IFS= read -r copy; do + found=$((found + 1)) + base="$(basename "$copy")" + + # Matched by basename across wit/, not by a constructed path: the canonical + # file is wit/tasks/tasks.wit, not wit/tasks.wit, and hard-coding either + # shape breaks on whichever one comes next. + mapfile -t canonical < <(find wit -type f -name "$base" | sort) + case "${#canonical[@]}" in + 1) ;; + 0) + echo "no canonical WIT for $copy (looked for $base under wit/)" >&2 + status=1 + continue + ;; + *) + echo "ambiguous: $base exists at ${canonical[*]}" >&2 + status=1 + continue + ;; + esac + + if cmp -s "$copy" "${canonical[0]}"; then + echo "ok $copy == ${canonical[0]}" + else + echo "DRIFT $copy differs from ${canonical[0]}" >&2 + diff -u "${canonical[0]}" "$copy" >&2 || true + status=1 + fi +done < <(find examples -type f -path '*/wit/deps/*.wit' -not -path '*/target/*' | sort) + +# A check that silently covers nothing is worse than no check, because it still +# reports green. If the vendored copies move, this says so instead. +if (( found == 0 )); then + echo "no vendored WIT found under examples/*/wit/deps/ — this check has stopped checking anything" >&2 + exit 1 +fi + +exit "$status" diff --git a/wit/notify/notify.wit b/wit/notify/notify.wit new file mode 100644 index 0000000..0edc9f8 --- /dev/null +++ b/wit/notify/notify.wit @@ -0,0 +1,40 @@ +package enclave:notify@0.1.0; + +/// Waking a tenant's devices. Calls are bound to the current tenant by the +/// runtime; no function here accepts a tenant id. +/// +/// **Nothing here carries text a person will read.** The runtime sends a +/// data-only message containing an opaque category and an optional +/// tenant-local reference, and nothing else. That payload crosses the parent +/// instance and Google, which are the two parties this design excludes from a +/// tenant's data — so the app wakes and fetches the detail over the attested +/// channel. See docs/NOTIFICATIONS.md. +interface notify { + /// Enrol an FCM registration token for this tenant. + /// + /// **Interactive only.** Enrolling grants a standing ability to reach a + /// device, and background work cannot grant itself that — the same rule + /// enclave:tasks/queue applies to enqueue. Enrolling a token this tenant + /// already has is success and changes nothing. + /// + /// 32-512 characters of [A-Za-z0-9_:.-]. + register-device: func(token: string) -> result<_, string>; + + /// Interactive only. Forgetting a token this tenant does not have is + /// success: a client pruned while it was offline is not wrong to ask. + forget-device: func(token: string) -> result<_, string>; + + /// How many devices this tenant has enrolled. Never the tokens themselves. + devices: func() -> result; + + /// Queue a wake signal to every device this tenant has enrolled. + /// + /// Allowed from background work, and that is the point: a finished task + /// telling its owner to come and look is the primary use. It authorizes + /// nothing — it spends an enrolment an interactive call already made. + /// + /// `category` and `reference` are opaque labels: 1-64 characters of + /// [A-Za-z0-9_-]. Not prose, not a message, not a title. Returns once the + /// signal is queued; delivery is best effort and unacknowledged. + wake: func(category: string, reference: option) -> result<_, string>; +} diff --git a/wit/stream/stream.wit b/wit/stream/stream.wit new file mode 100644 index 0000000..454316b --- /dev/null +++ b/wit/stream/stream.wit @@ -0,0 +1,96 @@ +package enclave:streams@0.1.0; + +/// A durable outbound connection the RUNTIME holds on a guest's behalf. +/// +/// # Why the runtime holds it and not the guest +/// +/// A guest has no execution context of its own. An instance is created to serve +/// one invocation and dropped at the end, so it cannot own a socket that +/// outlives the call — and nothing inside it can reconnect, because between +/// invocations there is no "inside it". +/// +/// That left exactly two ways for a guest to be reached: an inbound request, +/// which the gate binds to a passkey, and the task queue, which is a scheduler. +/// Neither serves a counterparty that wants to *initiate*. A service holding +/// half of an escrow key has no passkey for its user's tenant and cannot be +/// made to wait for a timer to come round. +/// +/// So the runtime keeps the connection, and the guest stays stateless: each +/// message the far side sends becomes one invocation, exactly as a task run is +/// one invocation. The guest replies from within that call, and may also send +/// unprompted from any invocation it is already in — an inbound request from +/// its owner, or another message. +/// +/// # What it is, on the wire +/// +/// Server-sent events for what arrives, POST for what is sent. Two HTTP shapes +/// rather than one socket, because that is what the egress path already +/// carries and what a service can serve without a WebSocket stack. The logical +/// channel is bidirectional; the transport is not exotic. +/// +/// ```text +/// GET /escrow/stream?id=- held open, text/event-stream +/// POST /escrow/send?id=- one message, and one reply +/// ``` +/// +/// Because what is sent is a request of its own and not a write into the held +/// one, the id is the ONLY thing tying the two together — which is why it names +/// the tenant as well as the stream. A far side serving several customers keys +/// its connections by the whole of it. +/// +/// # What is NOT promised +/// +/// Delivery is at least once in both directions, and a message may be seen +/// twice after a reconnect. Handlers deduplicate by their own identifiers, as +/// task handlers deduplicate by run id. Ordering holds within one connection +/// and not across a reconnect. +/// Names carry a `stream-` prefix although the interface already namespaces them. A guest mirrors +/// these onto a FLAT trait, and `connection.status` beside `queue.status` would be two different +/// things with one name on it — the sort of collision a mirror check cannot see through. +interface connection { + /// Ask the runtime to maintain a connection, and keep asking until told to + /// stop. Durable: it survives a runtime restart, and is re-established + /// without the guest being involved. + /// + /// `id` is a tenant-local idempotency key (ASCII letters, digits, '-' or + /// '_'). Opening one that already exists with the same origin is a no-op; + /// with a different origin it is an error rather than a silent move. + /// + /// **Tenant-local, and the far side is told so.** A guest usually derives + /// this from something about the counterparty, so every tenant it serves + /// opens under the same name. What goes on the wire is therefore the tenant + /// and this id together — a service that saw only this could not tell one + /// customer's connection from another's, nor tell which of them a message + /// sent to it belonged to, since that arrives as a request of its own. + /// + /// `origin` must be one the image allows, exactly as an outgoing request + /// must. A guest cannot reach further by asking the runtime to hold the + /// connection than by making the request itself. + stream-open: func(id: string, origin: string) -> result<_, string>; + + /// Stop maintaining it and forget the record. Idempotent. + stream-close: func(id: string) -> result<_, string>; + + /// Send one message to the far side. + /// + /// Fails if the connection is not currently established — the caller + /// decides whether that is fatal or worth repeating, because only the + /// caller knows whether the message is still wanted. + stream-send: func(id: string, payload: list) -> result<_, string>; + + /// A JSON record: whether it is connected, when it last was, and how many + /// times it has been re-established. + stream-status: func(id: string) -> result; +} + +world streaming { + import connection; + + /// One message from the far side, as one invocation. + /// + /// `message-id` is stable for a given message and repeats if it is + /// delivered again — deduplicate by it. A guest that returns an error has + /// the message redelivered; one that returns bytes has them sent back as + /// the reply, or none if it returns nothing to say. + export on-message: func(id: string, message-id: string, payload: list) -> result, string>; +} diff --git a/wit/tasks/tasks.wit b/wit/tasks/tasks.wit new file mode 100644 index 0000000..0b31db7 --- /dev/null +++ b/wit/tasks/tasks.wit @@ -0,0 +1,31 @@ +package enclave:tasks@0.1.0; + +/// Calls are bound to the current tenant by the runtime. Mutations require +/// an interactive invocation; a background invocation cannot grant itself work. +interface queue { + /// id is a tenant-local idempotency key (ASCII letters, digits, '-' or '_'). + /// run-at is Unix milliseconds. interval-ms, when present, is >= 1000. + enqueue: func(id: string, payload: list, run-at: u64, interval-ms: option) -> result<_, string>; + /// A JSON task record, including state, attempt count and result bytes. + status: func(id: string) -> result; + cancel: func(id: string) -> result<_, string>; + /// Remove a terminal record to release quota. Never removes running work. + forget: func(id: string) -> result<_, string>; +} + +world background { + import queue; + /// task-id is a run id, not the id the task was enqueued under. Its form + /// is `::`: the enqueued id, 32 lowercase hex + /// digits minted at enqueue, and a decimal count of recurring occurrences + /// starting at 0. An id cannot contain ':', so the enqueued id is + /// everything before the first one. + /// + /// It remains stable across retries of the same occurrence, and changes + /// for the next occurrence of a recurring task or for a task enqueued again + /// under the same id after `forget`. Deduplicate effects by the whole run + /// id; look up application state by the enqueued id. + /// + /// A successful result is persisted for the client; errors are retried. + export run-task: func(task-id: string, payload: list) -> result, string>; +} diff --git a/wit/world.wit b/wit/world.wit index 11affb5..3ff8ff5 100644 --- a/wit/world.wit +++ b/wit/world.wit @@ -2,7 +2,7 @@ // interfaces — wasi:io, wasi:clocks, etc. are reused from `wasmtime-wasi`. package s3fs:host; -world s3fs-host { +world enclave-runtime { import wasi:filesystem/types@0.2.6; import wasi:filesystem/preopens@0.2.6; }