diff --git a/Cargo.lock b/Cargo.lock index 6aba76e..3e53454 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -739,6 +739,7 @@ dependencies = [ "futures-util", "getrandom 0.3.4", "object_store", + "reqwest 0.12.28", "serde", "serde_json", "sysinfo", @@ -859,7 +860,7 @@ dependencies = [ [[package]] name = "cellule-app" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=70c3c218fafc815b951a40d4df1b50bba15b7d4c#70c3c218fafc815b951a40d4df1b50bba15b7d4c" +source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" dependencies = [ "blake3", "cellule-runtime", @@ -868,7 +869,7 @@ dependencies = [ [[package]] name = "cellule-host" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=70c3c218fafc815b951a40d4df1b50bba15b7d4c#70c3c218fafc815b951a40d4df1b50bba15b7d4c" +source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" dependencies = [ "cellule-app", "cellule-runtime", @@ -882,7 +883,7 @@ dependencies = [ [[package]] name = "cellule-ltx" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=70c3c218fafc815b951a40d4df1b50bba15b7d4c#70c3c218fafc815b951a40d4df1b50bba15b7d4c" +source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" dependencies = [ "async-trait", "blake3", @@ -904,7 +905,7 @@ dependencies = [ [[package]] name = "cellule-peer-http" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=70c3c218fafc815b951a40d4df1b50bba15b7d4c#70c3c218fafc815b951a40d4df1b50bba15b7d4c" +source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" dependencies = [ "axum", "cellule-runtime", @@ -926,7 +927,7 @@ dependencies = [ [[package]] name = "cellule-runtime" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=70c3c218fafc815b951a40d4df1b50bba15b7d4c#70c3c218fafc815b951a40d4df1b50bba15b7d4c" +source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" dependencies = [ "blake3", "bytes", @@ -952,7 +953,7 @@ dependencies = [ [[package]] name = "cellule-store" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=70c3c218fafc815b951a40d4df1b50bba15b7d4c#70c3c218fafc815b951a40d4df1b50bba15b7d4c" +source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" dependencies = [ "async-trait", "blake3", @@ -975,7 +976,7 @@ dependencies = [ [[package]] name = "cellule-types" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=70c3c218fafc815b951a40d4df1b50bba15b7d4c#70c3c218fafc815b951a40d4df1b50bba15b7d4c" +source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" dependencies = [ "schemars", "serde", diff --git a/Cargo.toml b/Cargo.toml index a14a629..fb94094 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -8,12 +8,12 @@ repository = "https://github.com/crabbuild/beyonddb" rust-version = "1.97" [workspace.dependencies] -cellule-app = { git = "https://github.com/crabbuild/cellule.git", rev = "70c3c218fafc815b951a40d4df1b50bba15b7d4c" } -cellule-host = { git = "https://github.com/crabbuild/cellule.git", rev = "70c3c218fafc815b951a40d4df1b50bba15b7d4c" } -cellule-peer-http = { git = "https://github.com/crabbuild/cellule.git", rev = "70c3c218fafc815b951a40d4df1b50bba15b7d4c" } -cellule-runtime = { git = "https://github.com/crabbuild/cellule.git", rev = "70c3c218fafc815b951a40d4df1b50bba15b7d4c" } -cellule-store = { git = "https://github.com/crabbuild/cellule.git", rev = "70c3c218fafc815b951a40d4df1b50bba15b7d4c" } -cellule-ltx = { git = "https://github.com/crabbuild/cellule.git", rev = "70c3c218fafc815b951a40d4df1b50bba15b7d4c" } +cellule-app = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } +cellule-host = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } +cellule-peer-http = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } +cellule-runtime = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } +cellule-store = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } +cellule-ltx = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } async-trait = "0.1" blake3 = "1.8" futures-util = "0.3" @@ -29,6 +29,7 @@ aws-credential-types = "1" aws-sdk-dynamodb = "1" ed25519-dalek = "2.2" object_store = { version = "0.14.1", default-features = false } +reqwest = { version = "0.12", default-features = false, features = ["rustls-tls", "stream"] } tempfile = "3" [package] @@ -60,6 +61,7 @@ extenddb-storage = { git = "https://github.com/crabbuild/extenddb.git", rev = "7 futures-util.workspace = true fs4 = "0.13.1" getrandom = "0.3" +reqwest.workspace = true serde.workspace = true serde_json.workspace = true sysinfo = { version = "0.38.4", default-features = false, features = ["system"] } diff --git a/README.md b/README.md index 7c0b7b6..4441d43 100644 --- a/README.md +++ b/README.md @@ -64,6 +64,12 @@ stream intent commit in one Cell command. The successful response follows durable publication. [PNG version](diagram/beyonddb-architecture/durable-write@2x.png) ยท [Detailed architecture and recovery design](docs/architecture.md) +The serving binary currently waits for object-store publication on each +durable write. Cellule's follower-log mode is not enabled in BeyondDB. An +opt-in persistent follower store and authenticated peer transport exist, but +the durability provider is not installed in serving and successor recovery +is unfinished. See the [follower durability design](docs/follower-durability.md). + ## Current capability boundary | Area | Current state | diff --git a/benchmarks/2026-09-29-cellule-main-attempt/README.md b/benchmarks/2026-09-29-cellule-main-attempt/README.md new file mode 100644 index 0000000..01181af --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/README.md @@ -0,0 +1,48 @@ +# Cellule `9e17746` release qualification attempt + +This is a **failed, incomplete performance qualification**, retained to make the +result reproducible. BeyondDB was built with Cellule +`9e17746a633ca1046bd93866866074226091f81a`; ExtendDB SQLite remained at +`7eaa89b437feed0af0f05883d3f1493f86c6fc6d`. The BeyondDB binary SHA-256 +and the exact pinned RustFS image are in [meta.json](meta.json). Both backends +used signed boto3 requests, a 1 KiB payload, 5 seconds per case, and one/eight +clients. BeyondDB used four initial partitions, 12 SQL workers, and the auth +cache. SQLite used its local file backend. See the fixture details in +[meta.json](meta.json) and [sqlite-meta.json](sqlite-meta.json). + +The host had 12 logical CPUs. Load was **44.5** at pair start, **47.5** after +BeyondDB, and **90.4** after SQLite. Other builds and virtual machines were +active. These numbers cannot establish an intrinsic throughput difference or +the impact of the Cellule update. + +| API | BeyondDB 1 / 8 clients (req/s) | SQLite 1 / 8 clients (req/s) | +| --- | ---: | ---: | +| GetItem | 125.85 / 283.41 | 260.56 / 482.32 | +| Query | 153.15 / 228.51 | 224.34 / 538.40 | +| Scan | 107.93 / 86.75 | 215.92 / 475.59 | +| BatchGetItem | 30.78 / 191.86 | 115.40 / 353.96 | +| TransactGetItems | 1.41 / 2.14 | 76.73 / 225.91 | + +The eight-client TransactGetItems case completed 21 requests and returned one +`TransactionCanceledException` with a `ThrottlingError` cancellation reason. +The BeyondDB server also logged deferred GSI projection and capacity sweeps, +incomplete cross-Cell participant resolution, and a transaction participant +capacity refusal from Cell mailbox-byte exhaustion. Zero errors in the first +nine cases therefore do not mean background work settled. +The fail-fast harness then stopped, leaving **10 of 24 BeyondDB cases** in +[run.json](run.json). SQLite completed **24 of 24 cases with zero request +errors** in [sqlite-run.json](sqlite-run.json). Full p50/p95/p99 latency, +elapsed time, and error fields are in those raw files. The harness now offers +`--continue-on-error` so the next overloaded attempt can collect all cases and +still exit unsuccessfully when any request fails. + +The separate signed release smoke check wrote a 1 KiB item, killed the serving +process, and read the same item after restart. All 17 library tests, formatting, +and strict Clippy passed. The broader ignored SDK restart test failed earlier +in GSI setup with HTTP 503 after Cell SQL deadlines on a host with load near +100. That failure remains unresolved. Neither this attempt nor the smoke check +proves the all-API SQLite performance objective. + +Run control, server warnings, and raw case output are in +[pair-status.json](pair-status.json), [server.log](server.log), and +[bench.log](bench.log). diff --git a/benchmarks/2026-09-29-cellule-main-attempt/bench.log b/benchmarks/2026-09-29-cellule-main-attempt/bench.log new file mode 100644 index 0000000..b031969 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/bench.log @@ -0,0 +1,11 @@ +{"operation": "get", "items_per_request": 1, "clients": 1, "elapsed_s": 5.01, "completed": 630, "errors": 0, "first_error": null, "ops_per_s": 125.85, "items_per_s": 125.85, "mean_ms": 7.94, "p50_ms": 4.93, "p95_ms": 18.72, "p99_ms": 26.41} +{"operation": "get", "items_per_request": 1, "clients": 8, "elapsed_s": 5.01, "completed": 1421, "errors": 0, "first_error": null, "ops_per_s": 283.41, "items_per_s": 283.41, "mean_ms": 28.11, "p50_ms": 21.15, "p95_ms": 64.72, "p99_ms": 120.36} +{"operation": "query", "items_per_request": 1, "clients": 1, "elapsed_s": 5.05, "completed": 773, "errors": 0, "first_error": null, "ops_per_s": 153.15, "items_per_s": 153.15, "mean_ms": 6.53, "p50_ms": 4.61, "p95_ms": 16.61, "p99_ms": 28.73} +{"operation": "query", "items_per_request": 1, "clients": 8, "elapsed_s": 5.01, "completed": 1146, "errors": 0, "first_error": null, "ops_per_s": 228.51, "items_per_s": 228.51, "mean_ms": 34.9, "p50_ms": 25.67, "p95_ms": 94.88, "p99_ms": 210.2} +{"operation": "scan", "items_per_request": 1, "clients": 1, "elapsed_s": 5.0, "completed": 540, "errors": 0, "first_error": null, "ops_per_s": 107.93, "items_per_s": 107.93, "mean_ms": 9.26, "p50_ms": 5.22, "p95_ms": 25.83, "p99_ms": 77.64} +{"operation": "scan", "items_per_request": 1, "clients": 8, "elapsed_s": 5.05, "completed": 438, "errors": 0, "first_error": null, "ops_per_s": 86.75, "items_per_s": 86.75, "mean_ms": 91.6, "p50_ms": 31.97, "p95_ms": 92.1, "p99_ms": 2885.71} +{"operation": "batch_get", "items_per_request": 2, "clients": 1, "elapsed_s": 5.0, "completed": 154, "errors": 0, "first_error": null, "ops_per_s": 30.78, "items_per_s": 61.56, "mean_ms": 32.45, "p50_ms": 20.11, "p95_ms": 109.1, "p99_ms": 202.28} +{"operation": "batch_get", "items_per_request": 2, "clients": 8, "elapsed_s": 5.03, "completed": 965, "errors": 0, "first_error": null, "ops_per_s": 191.86, "items_per_s": 383.72, "mean_ms": 41.49, "p50_ms": 29.77, "p95_ms": 94.39, "p99_ms": 225.75} +{"operation": "transact_get", "items_per_request": 2, "clients": 1, "elapsed_s": 5.67, "completed": 8, "errors": 0, "first_error": null, "ops_per_s": 1.41, "items_per_s": 2.82, "mean_ms": 708.14, "p50_ms": 852.61, "p95_ms": 1047.76, "p99_ms": 1047.76} +{"operation": "transact_get", "items_per_request": 2, "clients": 8, "elapsed_s": 9.83, "completed": 21, "errors": 1, "first_error": "TransactionCanceledException: An error occurred (TransactionCanceledException) when calling the TransactGetItems operation: Transaction cancelled, please refer cancellation reasons for specific reasons [ThrottlingError, None]", "ops_per_s": 2.14, "items_per_s": 4.27, "mean_ms": 3127.2, "p50_ms": 3610.88, "p95_ms": 5749.02, "p99_ms": 5749.02} +stopped after request error diff --git a/benchmarks/2026-09-29-cellule-main-attempt/beyond-runner.log b/benchmarks/2026-09-29-cellule-main-attempt/beyond-runner.log new file mode 100644 index 0000000..fc324d5 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/beyond-runner.log @@ -0,0 +1,2 @@ +run started; host load: (46.8740234375, 58.19580078125, 53.64794921875) +run finished exit: 1 host load: (47.5185546875, 56.4794921875, 53.38232421875) diff --git a/benchmarks/2026-09-29-cellule-main-attempt/meta.json b/benchmarks/2026-09-29-cellule-main-attempt/meta.json new file mode 100644 index 0000000..b78da9a --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/meta.json @@ -0,0 +1,29 @@ +{ + "server_commit": "ad4af2efd775409c2820a19013e59c79807d7c20 + Cellule 9e17746 pin and recovery fix", + "binary_sha256": "6429879d281eb49c9dd5b9b361007d52002560e64ce43a5da5c5a34e05c12abf", + "rustfs_image": "ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858", + "host_load_start": [ + 46.8740234375, + 58.19580078125, + 53.64794921875 + ], + "logical_cpus": 12, + "boto3_version": "1.43.105", + "fixture": { + "initial_partitions": 4, + "sql_workers": 12, + "max_active_cells": 128, + "follower_store_bytes": 1073741824, + "auth_cache_enabled": true, + "server_object_publication": true + }, + "endpoint": "http://127.0.0.1:53167", + "started_at_unix": 1790749922.373023, + "host_load_end": [ + 47.5185546875, + 56.4794921875, + 53.38232421875 + ], + "ended_at_unix": 1790749990.2865338, + "bench_exit_code": 1 +} diff --git a/benchmarks/2026-09-29-cellule-main-attempt/pair-status.json b/benchmarks/2026-09-29-cellule-main-attempt/pair-status.json new file mode 100644 index 0000000..b9c08ce --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/pair-status.json @@ -0,0 +1,22 @@ +{ + "root": "/var/folders/rn/k2b7mvbs2dgb__6ys6_h94pm0000gn/T/beyonddb-perf-current-zsbtxro3", + "started_at_unix": 1790749916.511093, + "host_load_start": [ + 44.51318359375, + 57.92919921875, + 53.5283203125 + ], + "beyond_exit_code": 1, + "host_load_after_beyond": [ + 47.5185546875, + 56.4794921875, + 53.38232421875 + ], + "sqlite_exit_code": 0, + "ended_at_unix": 1790750119.0698938, + "host_load_end": [ + 90.37353515625, + 66.35498046875, + 57.46142578125 + ] +} diff --git a/benchmarks/2026-09-29-cellule-main-attempt/run.json b/benchmarks/2026-09-29-cellule-main-attempt/run.json new file mode 100644 index 0000000..0c419a6 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/run.json @@ -0,0 +1,177 @@ +{ + "started_at": "2026-09-30T06:32:14.373084+00:00", + "endpoint": "http://127.0.0.1:53167", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 630, + "errors": 0, + "first_error": null, + "ops_per_s": 125.85, + "items_per_s": 125.85, + "mean_ms": 7.94, + "p50_ms": 4.93, + "p95_ms": 18.72, + "p99_ms": 26.41 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 1421, + "errors": 0, + "first_error": null, + "ops_per_s": 283.41, + "items_per_s": 283.41, + "mean_ms": 28.11, + "p50_ms": 21.15, + "p95_ms": 64.72, + "p99_ms": 120.36 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.05, + "completed": 773, + "errors": 0, + "first_error": null, + "ops_per_s": 153.15, + "items_per_s": 153.15, + "mean_ms": 6.53, + "p50_ms": 4.61, + "p95_ms": 16.61, + "p99_ms": 28.73 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 1146, + "errors": 0, + "first_error": null, + "ops_per_s": 228.51, + "items_per_s": 228.51, + "mean_ms": 34.9, + "p50_ms": 25.67, + "p95_ms": 94.88, + "p99_ms": 210.2 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 540, + "errors": 0, + "first_error": null, + "ops_per_s": 107.93, + "items_per_s": 107.93, + "mean_ms": 9.26, + "p50_ms": 5.22, + "p95_ms": 25.83, + "p99_ms": 77.64 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.05, + "completed": 438, + "errors": 0, + "first_error": null, + "ops_per_s": 86.75, + "items_per_s": 86.75, + "mean_ms": 91.6, + "p50_ms": 31.97, + "p95_ms": 92.1, + "p99_ms": 2885.71 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 154, + "errors": 0, + "first_error": null, + "ops_per_s": 30.78, + "items_per_s": 61.56, + "mean_ms": 32.45, + "p50_ms": 20.11, + "p95_ms": 109.1, + "p99_ms": 202.28 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.03, + "completed": 965, + "errors": 0, + "first_error": null, + "ops_per_s": 191.86, + "items_per_s": 383.72, + "mean_ms": 41.49, + "p50_ms": 29.77, + "p95_ms": 94.39, + "p99_ms": 225.75 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.67, + "completed": 8, + "errors": 0, + "first_error": null, + "ops_per_s": 1.41, + "items_per_s": 2.82, + "mean_ms": 708.14, + "p50_ms": 852.61, + "p95_ms": 1047.76, + "p99_ms": 1047.76 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 9.83, + "completed": 21, + "errors": 1, + "first_error": "TransactionCanceledException: An error occurred (TransactionCanceledException) when calling the TransactGetItems operation: Transaction cancelled, please refer cancellation reasons for specific reasons [ThrottlingError, None]", + "ops_per_s": 2.14, + "items_per_s": 4.27, + "mean_ms": 3127.2, + "p50_ms": 3610.88, + "p95_ms": 5749.02, + "p99_ms": 5749.02 + } + ] +} diff --git a/benchmarks/2026-09-29-cellule-main-attempt/server.log b/benchmarks/2026-09-29-cellule-main-attempt/server.log new file mode 100644 index 0000000..9e32f73 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/server.log @@ -0,0 +1,6 @@ +2026-09-30T06:32:50.770967Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T06:32:51.239427Z WARN beyonddb::provision::capacity: capacity sweep deferred account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T06:32:54.162214Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T06:32:57.275346Z WARN beyonddb::provision::transactions: transaction recovery deferred error=cross-Cell participant resolution is incomplete +2026-09-30T06:33:01.628778Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T06:33:07.977511Z WARN beyonddb::backend::transaction: transaction participant capacity refused error=Cell runtime capacity exhausted: Cell mailbox bytes diff --git a/benchmarks/2026-09-29-cellule-main-attempt/sqlite-bench.log b/benchmarks/2026-09-29-cellule-main-attempt/sqlite-bench.log new file mode 100644 index 0000000..77c5e72 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/sqlite-bench.log @@ -0,0 +1,24 @@ +{"operation": "get", "items_per_request": 1, "clients": 1, "elapsed_s": 5.01, "completed": 1305, "errors": 0, "first_error": null, "ops_per_s": 260.56, "items_per_s": 260.56, "mean_ms": 3.83, "p50_ms": 2.95, "p95_ms": 9.16, "p99_ms": 14.79} +{"operation": "get", "items_per_request": 1, "clients": 8, "elapsed_s": 5.01, "completed": 2415, "errors": 0, "first_error": null, "ops_per_s": 482.32, "items_per_s": 482.32, "mean_ms": 16.55, "p50_ms": 14.44, "p95_ms": 35.12, "p99_ms": 45.81} +{"operation": "query", "items_per_request": 1, "clients": 1, "elapsed_s": 5.0, "completed": 1122, "errors": 0, "first_error": null, "ops_per_s": 224.34, "items_per_s": 224.34, "mean_ms": 4.45, "p50_ms": 3.4, "p95_ms": 11.06, "p99_ms": 16.82} +{"operation": "query", "items_per_request": 1, "clients": 8, "elapsed_s": 5.01, "completed": 2695, "errors": 0, "first_error": null, "ops_per_s": 538.4, "items_per_s": 538.4, "mean_ms": 14.83, "p50_ms": 12.48, "p95_ms": 32.02, "p99_ms": 46.79} +{"operation": "scan", "items_per_request": 1, "clients": 1, "elapsed_s": 5.0, "completed": 1080, "errors": 0, "first_error": null, "ops_per_s": 215.92, "items_per_s": 215.92, "mean_ms": 4.63, "p50_ms": 3.68, "p95_ms": 10.77, "p99_ms": 16.58} +{"operation": "scan", "items_per_request": 1, "clients": 8, "elapsed_s": 5.01, "completed": 2383, "errors": 0, "first_error": null, "ops_per_s": 475.59, "items_per_s": 475.59, "mean_ms": 16.78, "p50_ms": 14.65, "p95_ms": 35.41, "p99_ms": 51.04} +{"operation": "batch_get", "items_per_request": 2, "clients": 1, "elapsed_s": 5.01, "completed": 578, "errors": 0, "first_error": null, "ops_per_s": 115.4, "items_per_s": 230.8, "mean_ms": 8.66, "p50_ms": 6.5, "p95_ms": 21.9, "p99_ms": 31.3} +{"operation": "batch_get", "items_per_request": 2, "clients": 8, "elapsed_s": 5.02, "completed": 1777, "errors": 0, "first_error": null, "ops_per_s": 353.96, "items_per_s": 707.91, "mean_ms": 22.53, "p50_ms": 19.66, "p95_ms": 46.97, "p99_ms": 62.07} +{"operation": "transact_get", "items_per_request": 2, "clients": 1, "elapsed_s": 5.03, "completed": 386, "errors": 0, "first_error": null, "ops_per_s": 76.73, "items_per_s": 153.45, "mean_ms": 13.03, "p50_ms": 10.85, "p95_ms": 29.06, "p99_ms": 47.62} +{"operation": "transact_get", "items_per_request": 2, "clients": 8, "elapsed_s": 5.02, "completed": 1135, "errors": 0, "first_error": null, "ops_per_s": 225.91, "items_per_s": 451.81, "mean_ms": 35.32, "p50_ms": 28.27, "p95_ms": 87.58, "p99_ms": 118.67} +{"operation": "describe_table", "items_per_request": 0, "clients": 1, "elapsed_s": 5.03, "completed": 574, "errors": 0, "first_error": null, "ops_per_s": 114.16, "items_per_s": 0.0, "mean_ms": 8.75, "p50_ms": 6.36, "p95_ms": 23.07, "p99_ms": 34.56} +{"operation": "describe_table", "items_per_request": 0, "clients": 8, "elapsed_s": 5.02, "completed": 1873, "errors": 0, "first_error": null, "ops_per_s": 373.33, "items_per_s": 0.0, "mean_ms": 21.38, "p50_ms": 18.27, "p95_ms": 46.42, "p99_ms": 66.2} +{"operation": "list_tables", "items_per_request": 0, "clients": 1, "elapsed_s": 5.0, "completed": 1214, "errors": 0, "first_error": null, "ops_per_s": 242.74, "items_per_s": 0.0, "mean_ms": 4.12, "p50_ms": 3.13, "p95_ms": 10.51, "p99_ms": 15.48} +{"operation": "list_tables", "items_per_request": 0, "clients": 8, "elapsed_s": 5.01, "completed": 2074, "errors": 0, "first_error": null, "ops_per_s": 413.75, "items_per_s": 0.0, "mean_ms": 19.31, "p50_ms": 15.47, "p95_ms": 46.38, "p99_ms": 87.43} +{"operation": "put", "items_per_request": 1, "clients": 1, "elapsed_s": 5.08, "completed": 236, "errors": 0, "first_error": null, "ops_per_s": 46.42, "items_per_s": 46.42, "mean_ms": 21.53, "p50_ms": 16.69, "p95_ms": 53.79, "p99_ms": 84.68} +{"operation": "put", "items_per_request": 1, "clients": 8, "elapsed_s": 5.12, "completed": 475, "errors": 0, "first_error": null, "ops_per_s": 92.71, "items_per_s": 92.71, "mean_ms": 85.86, "p50_ms": 25.35, "p95_ms": 333.82, "p99_ms": 443.11} +{"operation": "update", "items_per_request": 1, "clients": 1, "elapsed_s": 5.0, "completed": 218, "errors": 0, "first_error": null, "ops_per_s": 43.58, "items_per_s": 43.58, "mean_ms": 22.94, "p50_ms": 8.59, "p95_ms": 86.85, "p99_ms": 111.2} +{"operation": "update", "items_per_request": 1, "clients": 8, "elapsed_s": 5.11, "completed": 623, "errors": 0, "first_error": null, "ops_per_s": 121.93, "items_per_s": 121.93, "mean_ms": 65.39, "p50_ms": 34.41, "p95_ms": 239.86, "p99_ms": 327.88} +{"operation": "delete_missing", "items_per_request": 1, "clients": 1, "elapsed_s": 5.1, "completed": 408, "errors": 0, "first_error": null, "ops_per_s": 79.93, "items_per_s": 79.93, "mean_ms": 12.51, "p50_ms": 5.03, "p95_ms": 54.93, "p99_ms": 97.68} +{"operation": "delete_missing", "items_per_request": 1, "clients": 8, "elapsed_s": 5.14, "completed": 1432, "errors": 0, "first_error": null, "ops_per_s": 278.62, "items_per_s": 278.62, "mean_ms": 28.37, "p50_ms": 16.65, "p95_ms": 96.78, "p99_ms": 162.48} +{"operation": "batch_write", "items_per_request": 2, "clients": 1, "elapsed_s": 5.03, "completed": 158, "errors": 0, "first_error": null, "ops_per_s": 31.38, "items_per_s": 62.76, "mean_ms": 31.84, "p50_ms": 17.57, "p95_ms": 116.57, "p99_ms": 155.29} +{"operation": "batch_write", "items_per_request": 2, "clients": 8, "elapsed_s": 5.03, "completed": 412, "errors": 0, "first_error": null, "ops_per_s": 81.91, "items_per_s": 163.82, "mean_ms": 97.36, "p50_ms": 55.8, "p95_ms": 331.14, "p99_ms": 508.07} +{"operation": "transact_write", "items_per_request": 2, "clients": 1, "elapsed_s": 5.08, "completed": 156, "errors": 0, "first_error": null, "ops_per_s": 30.74, "items_per_s": 61.48, "mean_ms": 32.5, "p50_ms": 16.9, "p95_ms": 114.35, "p99_ms": 141.5} +{"operation": "transact_write", "items_per_request": 2, "clients": 8, "elapsed_s": 5.04, "completed": 519, "errors": 0, "first_error": null, "ops_per_s": 102.95, "items_per_s": 205.9, "mean_ms": 77.37, "p50_ms": 60.79, "p95_ms": 196.88, "p99_ms": 324.53} diff --git a/benchmarks/2026-09-29-cellule-main-attempt/sqlite-meta.json b/benchmarks/2026-09-29-cellule-main-attempt/sqlite-meta.json new file mode 100644 index 0000000..b62e830 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/sqlite-meta.json @@ -0,0 +1,24 @@ +{ + "server_commit": "7eaa89b437feed0af0f05883d3f1493f86c6fc6d", + "binary_sha256": "dfc9cd868d2bc06b71fa9fbc460a074dca330cb8ceffaebf62d8d54bdb485e38", + "host_load_start": [ + 46.435546875, + 56.10595703125, + 53.2685546875 + ], + "logical_cpus": 12, + "boto3_version": "1.43.105", + "fixture": { + "sqlite_file": "/var/folders/rn/k2b7mvbs2dgb__6ys6_h94pm0000gn/T/extenddb-perf-rerun-3qb_uch2/data.sqlite", + "dev_mode_open_auth": true + }, + "endpoint": "http://127.0.0.1:63152", + "started_at_unix": 1790749994.4709032, + "host_load_end": [ + 90.37353515625, + 66.35498046875, + 57.46142578125 + ], + "ended_at_unix": 1790750118.2884412, + "bench_exit_code": 0 +} diff --git a/benchmarks/2026-09-29-cellule-main-attempt/sqlite-run.json b/benchmarks/2026-09-29-cellule-main-attempt/sqlite-run.json new file mode 100644 index 0000000..85dc841 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/sqlite-run.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T06:33:15.242267+00:00", + "endpoint": "http://127.0.0.1:63152", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 1305, + "errors": 0, + "first_error": null, + "ops_per_s": 260.56, + "items_per_s": 260.56, + "mean_ms": 3.83, + "p50_ms": 2.95, + "p95_ms": 9.16, + "p99_ms": 14.79 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2415, + "errors": 0, + "first_error": null, + "ops_per_s": 482.32, + "items_per_s": 482.32, + "mean_ms": 16.55, + "p50_ms": 14.44, + "p95_ms": 35.12, + "p99_ms": 45.81 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1122, + "errors": 0, + "first_error": null, + "ops_per_s": 224.34, + "items_per_s": 224.34, + "mean_ms": 4.45, + "p50_ms": 3.4, + "p95_ms": 11.06, + "p99_ms": 16.82 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2695, + "errors": 0, + "first_error": null, + "ops_per_s": 538.4, + "items_per_s": 538.4, + "mean_ms": 14.83, + "p50_ms": 12.48, + "p95_ms": 32.02, + "p99_ms": 46.79 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1080, + "errors": 0, + "first_error": null, + "ops_per_s": 215.92, + "items_per_s": 215.92, + "mean_ms": 4.63, + "p50_ms": 3.68, + "p95_ms": 10.77, + "p99_ms": 16.58 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2383, + "errors": 0, + "first_error": null, + "ops_per_s": 475.59, + "items_per_s": 475.59, + "mean_ms": 16.78, + "p50_ms": 14.65, + "p95_ms": 35.41, + "p99_ms": 51.04 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.01, + "completed": 578, + "errors": 0, + "first_error": null, + "ops_per_s": 115.4, + "items_per_s": 230.8, + "mean_ms": 8.66, + "p50_ms": 6.5, + "p95_ms": 21.9, + "p99_ms": 31.3 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.02, + "completed": 1777, + "errors": 0, + "first_error": null, + "ops_per_s": 353.96, + "items_per_s": 707.91, + "mean_ms": 22.53, + "p50_ms": 19.66, + "p95_ms": 46.97, + "p99_ms": 62.07 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.03, + "completed": 386, + "errors": 0, + "first_error": null, + "ops_per_s": 76.73, + "items_per_s": 153.45, + "mean_ms": 13.03, + "p50_ms": 10.85, + "p95_ms": 29.06, + "p99_ms": 47.62 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.02, + "completed": 1135, + "errors": 0, + "first_error": null, + "ops_per_s": 225.91, + "items_per_s": 451.81, + "mean_ms": 35.32, + "p50_ms": 28.27, + "p95_ms": 87.58, + "p99_ms": 118.67 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.03, + "completed": 574, + "errors": 0, + "first_error": null, + "ops_per_s": 114.16, + "items_per_s": 0.0, + "mean_ms": 8.75, + "p50_ms": 6.36, + "p95_ms": 23.07, + "p99_ms": 34.56 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.02, + "completed": 1873, + "errors": 0, + "first_error": null, + "ops_per_s": 373.33, + "items_per_s": 0.0, + "mean_ms": 21.38, + "p50_ms": 18.27, + "p95_ms": 46.42, + "p99_ms": 66.2 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1214, + "errors": 0, + "first_error": null, + "ops_per_s": 242.74, + "items_per_s": 0.0, + "mean_ms": 4.12, + "p50_ms": 3.13, + "p95_ms": 10.51, + "p99_ms": 15.48 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2074, + "errors": 0, + "first_error": null, + "ops_per_s": 413.75, + "items_per_s": 0.0, + "mean_ms": 19.31, + "p50_ms": 15.47, + "p95_ms": 46.38, + "p99_ms": 87.43 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.08, + "completed": 236, + "errors": 0, + "first_error": null, + "ops_per_s": 46.42, + "items_per_s": 46.42, + "mean_ms": 21.53, + "p50_ms": 16.69, + "p95_ms": 53.79, + "p99_ms": 84.68 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.12, + "completed": 475, + "errors": 0, + "first_error": null, + "ops_per_s": 92.71, + "items_per_s": 92.71, + "mean_ms": 85.86, + "p50_ms": 25.35, + "p95_ms": 333.82, + "p99_ms": 443.11 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 218, + "errors": 0, + "first_error": null, + "ops_per_s": 43.58, + "items_per_s": 43.58, + "mean_ms": 22.94, + "p50_ms": 8.59, + "p95_ms": 86.85, + "p99_ms": 111.2 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.11, + "completed": 623, + "errors": 0, + "first_error": null, + "ops_per_s": 121.93, + "items_per_s": 121.93, + "mean_ms": 65.39, + "p50_ms": 34.41, + "p95_ms": 239.86, + "p99_ms": 327.88 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.1, + "completed": 408, + "errors": 0, + "first_error": null, + "ops_per_s": 79.93, + "items_per_s": 79.93, + "mean_ms": 12.51, + "p50_ms": 5.03, + "p95_ms": 54.93, + "p99_ms": 97.68 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.14, + "completed": 1432, + "errors": 0, + "first_error": null, + "ops_per_s": 278.62, + "items_per_s": 278.62, + "mean_ms": 28.37, + "p50_ms": 16.65, + "p95_ms": 96.78, + "p99_ms": 162.48 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.03, + "completed": 158, + "errors": 0, + "first_error": null, + "ops_per_s": 31.38, + "items_per_s": 62.76, + "mean_ms": 31.84, + "p50_ms": 17.57, + "p95_ms": 116.57, + "p99_ms": 155.29 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.03, + "completed": 412, + "errors": 0, + "first_error": null, + "ops_per_s": 81.91, + "items_per_s": 163.82, + "mean_ms": 97.36, + "p50_ms": 55.8, + "p95_ms": 331.14, + "p99_ms": 508.07 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.08, + "completed": 156, + "errors": 0, + "first_error": null, + "ops_per_s": 30.74, + "items_per_s": 61.48, + "mean_ms": 32.5, + "p50_ms": 16.9, + "p95_ms": 114.35, + "p99_ms": 141.5 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.04, + "completed": 519, + "errors": 0, + "first_error": null, + "ops_per_s": 102.95, + "items_per_s": 205.9, + "mean_ms": 77.37, + "p50_ms": 60.79, + "p95_ms": 196.88, + "p99_ms": 324.53 + } + ] +} diff --git a/benchmarks/2026-09-29-cellule-main-attempt/sqlite-runner.log b/benchmarks/2026-09-29-cellule-main-attempt/sqlite-runner.log new file mode 100644 index 0000000..9be6c44 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-attempt/sqlite-runner.log @@ -0,0 +1,2 @@ +SQLite run started; host load: (46.435546875, 56.10595703125, 53.2685546875) +SQLite run finished exit: 0 host load: (90.37353515625, 66.35498046875, 57.46142578125) diff --git a/benchmarks/2026-09-29-cellule-main-rerun/README.md b/benchmarks/2026-09-29-cellule-main-rerun/README.md new file mode 100644 index 0000000..e683381 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/README.md @@ -0,0 +1,70 @@ +# Cellule `9e17746` release rerun + +This is a complete **local diagnostic run**, not a production capacity or +controlled speed-ratio claim. BeyondDB used a release binary built with +Cellule `9e17746a633ca1046bd93866866074226091f81a`; the comparison used +file-backed ExtendDB SQLite at `7eaa89b437feed0af0f05883d3f1493f86c6fc6d`. +The binaries' SHA-256 values are in [BeyondDB metadata](meta.json) and +[SQLite metadata](sqlite-meta.json). BeyondDB published to the pinned RustFS +image recorded in its metadata. + +Both fixtures used signed boto3 requests, 64 seeded items with 1 KiB payloads, +five seconds per case, one and eight clients, strong reads where the API +supports them, and SDK retries disabled. BeyondDB used four initial +partitions, 12 SQL workers, and `auth_cache_enabled: true`. The cases ran +sequentially on a 12-logical-CPU workstation. The host's one-minute load was +**47.5** at pair start, **30.3** after BeyondDB, and **37.3** after SQLite; +other builds and virtual machines were active. SQLite's local durability and +development authorization differ from BeyondDB's IAM checks and RustFS +publication. See [run control](pair-status.json) for exact times and load. + +## Requests per second + +| API | BeyondDB 1 | SQLite 1 | BeyondDB 8 | SQLite 8 | +| --- | ---: | ---: | ---: | ---: | +| GetItem | 61.01 | 279.56 | 235.04 | 507.24 | +| Query | 50.52 | 381.05 | 232.39 | 610.54 | +| Scan | 70.57 | 367.69 | 656.98 | 754.88 | +| BatchGetItem | 236.93 | 187.15 | 618.38 | 578.24 | +| TransactGetItems | 1.28 | 171.78 | 4.18 | 473.26 | +| DescribeTable | 491.35 | 258.86 | 493.66 | 408.94 | +| ListTables | 571.06 | 359.12 | 819.75 | 496.83 | +| PutItem | 22.57 | 92.52 | 77.61 | 54.71 | +| UpdateItem | 18.36 | 57.18 | 34.23 | 47.02 | +| DeleteItem, absent key | 13.04 | 85.32 | 15.93 | 246.95 | +| BatchWriteItem | 2.53 | 31.34 | 8.92 | 32.28 | +| TransactWriteItems | 0.30 | 16.97 | 1.50 | 131.07 | + +Batch and transaction cases carry two items per request. For example, +BeyondDB's eight-client BatchGetItem result is **1,236.76 items/s**. The raw +[BeyondDB](run.json) and [SQLite](sqlite-run.json) JSON contain item rates, +completed requests, elapsed times, and all latency percentiles. + +## p95 latency in milliseconds + +| API | BeyondDB 1 | SQLite 1 | BeyondDB 8 | SQLite 8 | +| --- | ---: | ---: | ---: | ---: | +| GetItem | 68.03 | 9.81 | 109.67 | 30.47 | +| Query | 88.26 | 5.52 | 129.83 | 29.65 | +| Scan | 44.03 | 6.65 | 24.54 | 25.21 | +| BatchGetItem | 7.40 | 12.96 | 24.04 | 28.67 | +| TransactGetItems | 1,148.83 | 18.75 | 2,714.87 | 33.01 | +| DescribeTable | 3.71 | 10.40 | 28.67 | 45.37 | +| ListTables | 3.32 | 5.47 | 20.97 | 35.28 | +| PutItem | 77.21 | 30.00 | 288.42 | 311.05 | +| UpdateItem | 94.44 | 58.88 | 516.49 | 380.32 | +| DeleteItem, absent key | 136.76 | 38.59 | 1,509.38 | 104.49 | +| BatchWriteItem | 722.80 | 78.07 | 1,668.79 | 329.08 | +| TransactWriteItems | 1,812.97 | 126.79 | 5,995.43 | 170.65 | + +All **24 BeyondDB** and **24 SQLite** cases completed with **zero foreground +SDK request errors**. BeyondDB's [server log](server.log) nevertheless records +two deferred capacity sweeps from Cell mailbox-byte exhaustion and one +deferred cross-Cell participant resolution. Foreground success does not prove +that background work settled. + +BeyondDB exceeded SQLite in BatchGetItem and metadata reads at both client +counts, and in PutItem at eight clients on this run. It remained far behind in +transactions and most writes. The changing external load prevents attributing +any rate difference to the Cellule pin. The all-API SQLite performance +objective is **not met**. diff --git a/benchmarks/2026-09-29-cellule-main-rerun/bench.log b/benchmarks/2026-09-29-cellule-main-rerun/bench.log new file mode 100644 index 0000000..1d59216 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/bench.log @@ -0,0 +1,24 @@ +{"operation": "get", "items_per_request": 1, "clients": 1, "elapsed_s": 5.03, "completed": 307, "errors": 0, "first_error": null, "ops_per_s": 61.01, "items_per_s": 61.01, "mean_ms": 16.38, "p50_ms": 3.89, "p95_ms": 68.03, "p99_ms": 143.01} +{"operation": "get", "items_per_request": 1, "clients": 8, "elapsed_s": 5.02, "completed": 1181, "errors": 0, "first_error": null, "ops_per_s": 235.04, "items_per_s": 235.04, "mean_ms": 33.89, "p50_ms": 18.79, "p95_ms": 109.67, "p99_ms": 183.5} +{"operation": "query", "items_per_request": 1, "clients": 1, "elapsed_s": 5.01, "completed": 253, "errors": 0, "first_error": null, "ops_per_s": 50.52, "items_per_s": 50.52, "mean_ms": 19.79, "p50_ms": 5.81, "p95_ms": 88.26, "p99_ms": 154.63} +{"operation": "query", "items_per_request": 1, "clients": 8, "elapsed_s": 5.07, "completed": 1178, "errors": 0, "first_error": null, "ops_per_s": 232.39, "items_per_s": 232.39, "mean_ms": 34.26, "p50_ms": 13.76, "p95_ms": 129.83, "p99_ms": 355.41} +{"operation": "scan", "items_per_request": 1, "clients": 1, "elapsed_s": 5.0, "completed": 353, "errors": 0, "first_error": null, "ops_per_s": 70.57, "items_per_s": 70.57, "mean_ms": 14.17, "p50_ms": 5.54, "p95_ms": 44.03, "p99_ms": 163.04} +{"operation": "scan", "items_per_request": 1, "clients": 8, "elapsed_s": 5.01, "completed": 3291, "errors": 0, "first_error": null, "ops_per_s": 656.98, "items_per_s": 656.98, "mean_ms": 12.16, "p50_ms": 10.1, "p95_ms": 24.54, "p99_ms": 59.01} +{"operation": "batch_get", "items_per_request": 2, "clients": 1, "elapsed_s": 5.0, "completed": 1185, "errors": 0, "first_error": null, "ops_per_s": 236.93, "items_per_s": 473.86, "mean_ms": 4.22, "p50_ms": 3.45, "p95_ms": 7.4, "p99_ms": 13.86} +{"operation": "batch_get", "items_per_request": 2, "clients": 8, "elapsed_s": 5.03, "completed": 3110, "errors": 0, "first_error": null, "ops_per_s": 618.38, "items_per_s": 1236.76, "mean_ms": 12.9, "p50_ms": 11.19, "p95_ms": 24.04, "p99_ms": 44.97} +{"operation": "transact_get", "items_per_request": 2, "clients": 1, "elapsed_s": 5.48, "completed": 7, "errors": 0, "first_error": null, "ops_per_s": 1.28, "items_per_s": 2.56, "mean_ms": 782.5, "p50_ms": 1011.6, "p95_ms": 1148.83, "p99_ms": 1148.83} +{"operation": "transact_get", "items_per_request": 2, "clients": 8, "elapsed_s": 6.23, "completed": 26, "errors": 0, "first_error": null, "ops_per_s": 4.18, "items_per_s": 8.35, "mean_ms": 1726.69, "p50_ms": 2033.59, "p95_ms": 2714.87, "p99_ms": 2954.75} +{"operation": "describe_table", "items_per_request": 0, "clients": 1, "elapsed_s": 5.0, "completed": 2458, "errors": 0, "first_error": null, "ops_per_s": 491.35, "items_per_s": 0.0, "mean_ms": 2.03, "p50_ms": 1.65, "p95_ms": 3.71, "p99_ms": 9.35} +{"operation": "describe_table", "items_per_request": 0, "clients": 8, "elapsed_s": 5.01, "completed": 2471, "errors": 0, "first_error": null, "ops_per_s": 493.66, "items_per_s": 0.0, "mean_ms": 16.17, "p50_ms": 10.14, "p95_ms": 28.67, "p99_ms": 49.94} +{"operation": "list_tables", "items_per_request": 0, "clients": 1, "elapsed_s": 5.0, "completed": 2857, "errors": 0, "first_error": null, "ops_per_s": 571.06, "items_per_s": 0.0, "mean_ms": 1.75, "p50_ms": 1.59, "p95_ms": 3.32, "p99_ms": 4.95} +{"operation": "list_tables", "items_per_request": 0, "clients": 8, "elapsed_s": 5.0, "completed": 4102, "errors": 0, "first_error": null, "ops_per_s": 819.75, "items_per_s": 0.0, "mean_ms": 9.75, "p50_ms": 8.46, "p95_ms": 20.97, "p99_ms": 29.85} +{"operation": "put", "items_per_request": 1, "clients": 1, "elapsed_s": 5.01, "completed": 113, "errors": 0, "first_error": null, "ops_per_s": 22.57, "items_per_s": 22.57, "mean_ms": 44.3, "p50_ms": 38.62, "p95_ms": 77.21, "p99_ms": 143.58} +{"operation": "put", "items_per_request": 1, "clients": 8, "elapsed_s": 5.01, "completed": 389, "errors": 0, "first_error": null, "ops_per_s": 77.61, "items_per_s": 77.61, "mean_ms": 102.99, "p50_ms": 81.64, "p95_ms": 288.42, "p99_ms": 416.12} +{"operation": "update", "items_per_request": 1, "clients": 1, "elapsed_s": 5.12, "completed": 94, "errors": 0, "first_error": null, "ops_per_s": 18.36, "items_per_s": 18.36, "mean_ms": 54.46, "p50_ms": 45.11, "p95_ms": 94.44, "p99_ms": 244.44} +{"operation": "update", "items_per_request": 1, "clients": 8, "elapsed_s": 5.29, "completed": 181, "errors": 0, "first_error": null, "ops_per_s": 34.23, "items_per_s": 34.23, "mean_ms": 226.65, "p50_ms": 178.16, "p95_ms": 516.49, "p99_ms": 634.56} +{"operation": "delete_missing", "items_per_request": 1, "clients": 1, "elapsed_s": 5.06, "completed": 66, "errors": 0, "first_error": null, "ops_per_s": 13.04, "items_per_s": 13.04, "mean_ms": 76.62, "p50_ms": 63.48, "p95_ms": 136.76, "p99_ms": 169.98} +{"operation": "delete_missing", "items_per_request": 1, "clients": 8, "elapsed_s": 7.72, "completed": 123, "errors": 0, "first_error": null, "ops_per_s": 15.93, "items_per_s": 15.93, "mean_ms": 395.18, "p50_ms": 189.41, "p95_ms": 1509.38, "p99_ms": 2286.07} +{"operation": "batch_write", "items_per_request": 2, "clients": 1, "elapsed_s": 5.14, "completed": 13, "errors": 0, "first_error": null, "ops_per_s": 2.53, "items_per_s": 5.06, "mean_ms": 395.45, "p50_ms": 318.02, "p95_ms": 722.8, "p99_ms": 722.8} +{"operation": "batch_write", "items_per_request": 2, "clients": 8, "elapsed_s": 6.28, "completed": 56, "errors": 0, "first_error": null, "ops_per_s": 8.92, "items_per_s": 17.84, "mean_ms": 758.5, "p50_ms": 612.26, "p95_ms": 1668.79, "p99_ms": 2185.26} +{"operation": "transact_write", "items_per_request": 2, "clients": 1, "elapsed_s": 6.77, "completed": 2, "errors": 0, "first_error": null, "ops_per_s": 0.3, "items_per_s": 0.59, "mean_ms": 3384.62, "p50_ms": 1812.97, "p95_ms": 1812.97, "p99_ms": 1812.97} +{"operation": "transact_write", "items_per_request": 2, "clients": 8, "elapsed_s": 8.03, "completed": 12, "errors": 0, "first_error": null, "ops_per_s": 1.5, "items_per_s": 2.99, "mean_ms": 4444.77, "p50_ms": 4287.26, "p95_ms": 5995.43, "p99_ms": 5995.43} diff --git a/benchmarks/2026-09-29-cellule-main-rerun/beyond-runner.log b/benchmarks/2026-09-29-cellule-main-rerun/beyond-runner.log new file mode 100644 index 0000000..2dac143 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/beyond-runner.log @@ -0,0 +1,2 @@ +run started; host load: (47.21875, 54.048828125, 56.50927734375) +run finished exit: 0 host load: (30.32177734375, 44.060546875, 51.97900390625) diff --git a/benchmarks/2026-09-29-cellule-main-rerun/meta.json b/benchmarks/2026-09-29-cellule-main-rerun/meta.json new file mode 100644 index 0000000..bf9cda1 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/meta.json @@ -0,0 +1,29 @@ +{ + "server_commit": "3c712f4526c3a59e361925f9df0872b2b681defa + Cellule 9e17746 pin and recovery fix", + "binary_sha256": "6429879d281eb49c9dd5b9b361007d52002560e64ce43a5da5c5a34e05c12abf", + "rustfs_image": "ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858", + "host_load_start": [ + 47.21875, + 54.048828125, + 56.50927734375 + ], + "logical_cpus": 12, + "boto3_version": "1.43.105", + "fixture": { + "initial_partitions": 4, + "sql_workers": 12, + "max_active_cells": 128, + "follower_store_bytes": 1073741824, + "auth_cache_enabled": true, + "server_object_publication": true + }, + "endpoint": "http://127.0.0.1:52340", + "started_at_unix": 1790751025.832394, + "host_load_end": [ + 30.32177734375, + 44.060546875, + 51.97900390625 + ], + "ended_at_unix": 1790751188.958745, + "bench_exit_code": 0 +} diff --git a/benchmarks/2026-09-29-cellule-main-rerun/pair-status.json b/benchmarks/2026-09-29-cellule-main-rerun/pair-status.json new file mode 100644 index 0000000..24f17c4 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/pair-status.json @@ -0,0 +1,22 @@ +{ + "root": "/var/folders/rn/k2b7mvbs2dgb__6ys6_h94pm0000gn/T/beyonddb-perf-current-e5u_kfni", + "started_at_unix": 1790751017.3765981, + "host_load_start": [ + 47.482421875, + 54.33837890625, + 56.6396484375 + ], + "beyond_exit_code": 0, + "host_load_after_beyond": [ + 30.32177734375, + 44.060546875, + 51.97900390625 + ], + "sqlite_exit_code": 0, + "ended_at_unix": 1790751317.781413, + "host_load_end": [ + 37.271484375, + 41.0068359375, + 49.60595703125 + ] +} diff --git a/benchmarks/2026-09-29-cellule-main-rerun/run.json b/benchmarks/2026-09-29-cellule-main-rerun/run.json new file mode 100644 index 0000000..80af3dd --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/run.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T06:50:57.302199+00:00", + "endpoint": "http://127.0.0.1:52340", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.03, + "completed": 307, + "errors": 0, + "first_error": null, + "ops_per_s": 61.01, + "items_per_s": 61.01, + "mean_ms": 16.38, + "p50_ms": 3.89, + "p95_ms": 68.03, + "p99_ms": 143.01 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.02, + "completed": 1181, + "errors": 0, + "first_error": null, + "ops_per_s": 235.04, + "items_per_s": 235.04, + "mean_ms": 33.89, + "p50_ms": 18.79, + "p95_ms": 109.67, + "p99_ms": 183.5 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 253, + "errors": 0, + "first_error": null, + "ops_per_s": 50.52, + "items_per_s": 50.52, + "mean_ms": 19.79, + "p50_ms": 5.81, + "p95_ms": 88.26, + "p99_ms": 154.63 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.07, + "completed": 1178, + "errors": 0, + "first_error": null, + "ops_per_s": 232.39, + "items_per_s": 232.39, + "mean_ms": 34.26, + "p50_ms": 13.76, + "p95_ms": 129.83, + "p99_ms": 355.41 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 353, + "errors": 0, + "first_error": null, + "ops_per_s": 70.57, + "items_per_s": 70.57, + "mean_ms": 14.17, + "p50_ms": 5.54, + "p95_ms": 44.03, + "p99_ms": 163.04 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3291, + "errors": 0, + "first_error": null, + "ops_per_s": 656.98, + "items_per_s": 656.98, + "mean_ms": 12.16, + "p50_ms": 10.1, + "p95_ms": 24.54, + "p99_ms": 59.01 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1185, + "errors": 0, + "first_error": null, + "ops_per_s": 236.93, + "items_per_s": 473.86, + "mean_ms": 4.22, + "p50_ms": 3.45, + "p95_ms": 7.4, + "p99_ms": 13.86 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.03, + "completed": 3110, + "errors": 0, + "first_error": null, + "ops_per_s": 618.38, + "items_per_s": 1236.76, + "mean_ms": 12.9, + "p50_ms": 11.19, + "p95_ms": 24.04, + "p99_ms": 44.97 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.48, + "completed": 7, + "errors": 0, + "first_error": null, + "ops_per_s": 1.28, + "items_per_s": 2.56, + "mean_ms": 782.5, + "p50_ms": 1011.6, + "p95_ms": 1148.83, + "p99_ms": 1148.83 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 6.23, + "completed": 26, + "errors": 0, + "first_error": null, + "ops_per_s": 4.18, + "items_per_s": 8.35, + "mean_ms": 1726.69, + "p50_ms": 2033.59, + "p95_ms": 2714.87, + "p99_ms": 2954.75 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2458, + "errors": 0, + "first_error": null, + "ops_per_s": 491.35, + "items_per_s": 0.0, + "mean_ms": 2.03, + "p50_ms": 1.65, + "p95_ms": 3.71, + "p99_ms": 9.35 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2471, + "errors": 0, + "first_error": null, + "ops_per_s": 493.66, + "items_per_s": 0.0, + "mean_ms": 16.17, + "p50_ms": 10.14, + "p95_ms": 28.67, + "p99_ms": 49.94 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2857, + "errors": 0, + "first_error": null, + "ops_per_s": 571.06, + "items_per_s": 0.0, + "mean_ms": 1.75, + "p50_ms": 1.59, + "p95_ms": 3.32, + "p99_ms": 4.95 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.0, + "completed": 4102, + "errors": 0, + "first_error": null, + "ops_per_s": 819.75, + "items_per_s": 0.0, + "mean_ms": 9.75, + "p50_ms": 8.46, + "p95_ms": 20.97, + "p99_ms": 29.85 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 113, + "errors": 0, + "first_error": null, + "ops_per_s": 22.57, + "items_per_s": 22.57, + "mean_ms": 44.3, + "p50_ms": 38.62, + "p95_ms": 77.21, + "p99_ms": 143.58 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 389, + "errors": 0, + "first_error": null, + "ops_per_s": 77.61, + "items_per_s": 77.61, + "mean_ms": 102.99, + "p50_ms": 81.64, + "p95_ms": 288.42, + "p99_ms": 416.12 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.12, + "completed": 94, + "errors": 0, + "first_error": null, + "ops_per_s": 18.36, + "items_per_s": 18.36, + "mean_ms": 54.46, + "p50_ms": 45.11, + "p95_ms": 94.44, + "p99_ms": 244.44 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.29, + "completed": 181, + "errors": 0, + "first_error": null, + "ops_per_s": 34.23, + "items_per_s": 34.23, + "mean_ms": 226.65, + "p50_ms": 178.16, + "p95_ms": 516.49, + "p99_ms": 634.56 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.06, + "completed": 66, + "errors": 0, + "first_error": null, + "ops_per_s": 13.04, + "items_per_s": 13.04, + "mean_ms": 76.62, + "p50_ms": 63.48, + "p95_ms": 136.76, + "p99_ms": 169.98 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 7.72, + "completed": 123, + "errors": 0, + "first_error": null, + "ops_per_s": 15.93, + "items_per_s": 15.93, + "mean_ms": 395.18, + "p50_ms": 189.41, + "p95_ms": 1509.38, + "p99_ms": 2286.07 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.14, + "completed": 13, + "errors": 0, + "first_error": null, + "ops_per_s": 2.53, + "items_per_s": 5.06, + "mean_ms": 395.45, + "p50_ms": 318.02, + "p95_ms": 722.8, + "p99_ms": 722.8 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 6.28, + "completed": 56, + "errors": 0, + "first_error": null, + "ops_per_s": 8.92, + "items_per_s": 17.84, + "mean_ms": 758.5, + "p50_ms": 612.26, + "p95_ms": 1668.79, + "p99_ms": 2185.26 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 6.77, + "completed": 2, + "errors": 0, + "first_error": null, + "ops_per_s": 0.3, + "items_per_s": 0.59, + "mean_ms": 3384.62, + "p50_ms": 1812.97, + "p95_ms": 1812.97, + "p99_ms": 1812.97 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 8.03, + "completed": 12, + "errors": 0, + "first_error": null, + "ops_per_s": 1.5, + "items_per_s": 2.99, + "mean_ms": 4444.77, + "p50_ms": 4287.26, + "p95_ms": 5995.43, + "p99_ms": 5995.43 + } + ] +} diff --git a/benchmarks/2026-09-29-cellule-main-rerun/server.log b/benchmarks/2026-09-29-cellule-main-rerun/server.log new file mode 100644 index 0000000..7007354 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/server.log @@ -0,0 +1,3 @@ +2026-09-30T06:51:13.469384Z WARN beyonddb::provision::capacity: capacity sweep deferred account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T06:51:41.137538Z WARN beyonddb::provision::transactions: transaction recovery deferred error=cross-Cell participant resolution is incomplete +2026-09-30T06:52:28.392136Z WARN beyonddb::provision::capacity: capacity sweep deferred account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes diff --git a/benchmarks/2026-09-29-cellule-main-rerun/sqlite-bench.log b/benchmarks/2026-09-29-cellule-main-rerun/sqlite-bench.log new file mode 100644 index 0000000..9522418 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/sqlite-bench.log @@ -0,0 +1,24 @@ +{"operation": "get", "items_per_request": 1, "clients": 1, "elapsed_s": 5.0, "completed": 1398, "errors": 0, "first_error": null, "ops_per_s": 279.56, "items_per_s": 279.56, "mean_ms": 3.58, "p50_ms": 2.42, "p95_ms": 9.81, "p99_ms": 18.8} +{"operation": "get", "items_per_request": 1, "clients": 8, "elapsed_s": 5.01, "completed": 2541, "errors": 0, "first_error": null, "ops_per_s": 507.24, "items_per_s": 507.24, "mean_ms": 15.74, "p50_ms": 11.36, "p95_ms": 30.47, "p99_ms": 71.22} +{"operation": "query", "items_per_request": 1, "clients": 1, "elapsed_s": 5.0, "completed": 1907, "errors": 0, "first_error": null, "ops_per_s": 381.05, "items_per_s": 381.05, "mean_ms": 2.62, "p50_ms": 1.86, "p95_ms": 5.52, "p99_ms": 13.14} +{"operation": "query", "items_per_request": 1, "clients": 8, "elapsed_s": 5.01, "completed": 3057, "errors": 0, "first_error": null, "ops_per_s": 610.54, "items_per_s": 610.54, "mean_ms": 13.08, "p50_ms": 10.48, "p95_ms": 29.65, "p99_ms": 52.03} +{"operation": "scan", "items_per_request": 1, "clients": 1, "elapsed_s": 5.0, "completed": 1839, "errors": 0, "first_error": null, "ops_per_s": 367.69, "items_per_s": 367.69, "mean_ms": 2.72, "p50_ms": 2.1, "p95_ms": 6.65, "p99_ms": 11.03} +{"operation": "scan", "items_per_request": 1, "clients": 8, "elapsed_s": 5.0, "completed": 3776, "errors": 0, "first_error": null, "ops_per_s": 754.88, "items_per_s": 754.88, "mean_ms": 10.59, "p50_ms": 8.56, "p95_ms": 25.21, "p99_ms": 38.19} +{"operation": "batch_get", "items_per_request": 2, "clients": 1, "elapsed_s": 5.0, "completed": 936, "errors": 0, "first_error": null, "ops_per_s": 187.15, "items_per_s": 374.3, "mean_ms": 5.34, "p50_ms": 4.02, "p95_ms": 12.96, "p99_ms": 19.41} +{"operation": "batch_get", "items_per_request": 2, "clients": 8, "elapsed_s": 5.01, "completed": 2895, "errors": 0, "first_error": null, "ops_per_s": 578.24, "items_per_s": 1156.48, "mean_ms": 13.82, "p50_ms": 11.35, "p95_ms": 28.67, "p99_ms": 50.73} +{"operation": "transact_get", "items_per_request": 2, "clients": 1, "elapsed_s": 5.01, "completed": 861, "errors": 0, "first_error": null, "ops_per_s": 171.78, "items_per_s": 343.57, "mean_ms": 5.82, "p50_ms": 3.3, "p95_ms": 18.75, "p99_ms": 37.54} +{"operation": "transact_get", "items_per_request": 2, "clients": 8, "elapsed_s": 5.01, "completed": 2370, "errors": 0, "first_error": null, "ops_per_s": 473.26, "items_per_s": 946.52, "mean_ms": 16.87, "p50_ms": 14.16, "p95_ms": 33.01, "p99_ms": 57.61} +{"operation": "describe_table", "items_per_request": 0, "clients": 1, "elapsed_s": 5.01, "completed": 1296, "errors": 0, "first_error": null, "ops_per_s": 258.86, "items_per_s": 0.0, "mean_ms": 3.86, "p50_ms": 2.53, "p95_ms": 10.4, "p99_ms": 25.92} +{"operation": "describe_table", "items_per_request": 0, "clients": 8, "elapsed_s": 5.01, "completed": 2047, "errors": 0, "first_error": null, "ops_per_s": 408.94, "items_per_s": 0.0, "mean_ms": 19.52, "p50_ms": 15.54, "p95_ms": 45.37, "p99_ms": 81.94} +{"operation": "list_tables", "items_per_request": 0, "clients": 1, "elapsed_s": 5.0, "completed": 1796, "errors": 0, "first_error": null, "ops_per_s": 359.12, "items_per_s": 0.0, "mean_ms": 2.78, "p50_ms": 2.12, "p95_ms": 5.47, "p99_ms": 13.81} +{"operation": "list_tables", "items_per_request": 0, "clients": 8, "elapsed_s": 5.02, "completed": 2493, "errors": 0, "first_error": null, "ops_per_s": 496.83, "items_per_s": 0.0, "mean_ms": 16.05, "p50_ms": 11.48, "p95_ms": 35.28, "p99_ms": 126.37} +{"operation": "put", "items_per_request": 1, "clients": 1, "elapsed_s": 5.02, "completed": 464, "errors": 0, "first_error": null, "ops_per_s": 92.52, "items_per_s": 92.52, "mean_ms": 10.8, "p50_ms": 4.84, "p95_ms": 30.0, "p99_ms": 64.34} +{"operation": "put", "items_per_request": 1, "clients": 8, "elapsed_s": 5.01, "completed": 274, "errors": 0, "first_error": null, "ops_per_s": 54.71, "items_per_s": 54.71, "mean_ms": 145.99, "p50_ms": 79.97, "p95_ms": 311.05, "p99_ms": 1633.93} +{"operation": "update", "items_per_request": 1, "clients": 1, "elapsed_s": 5.0, "completed": 286, "errors": 0, "first_error": null, "ops_per_s": 57.18, "items_per_s": 57.18, "mean_ms": 17.49, "p50_ms": 8.24, "p95_ms": 58.88, "p99_ms": 113.28} +{"operation": "update", "items_per_request": 1, "clients": 8, "elapsed_s": 5.17, "completed": 243, "errors": 0, "first_error": null, "ops_per_s": 47.02, "items_per_s": 47.02, "mean_ms": 168.57, "p50_ms": 140.05, "p95_ms": 380.32, "p99_ms": 762.03} +{"operation": "delete_missing", "items_per_request": 1, "clients": 1, "elapsed_s": 5.0, "completed": 427, "errors": 0, "first_error": null, "ops_per_s": 85.32, "items_per_s": 85.32, "mean_ms": 11.71, "p50_ms": 6.38, "p95_ms": 38.59, "p99_ms": 70.44} +{"operation": "delete_missing", "items_per_request": 1, "clients": 8, "elapsed_s": 5.03, "completed": 1242, "errors": 0, "first_error": null, "ops_per_s": 246.95, "items_per_s": 246.95, "mean_ms": 32.29, "p50_ms": 18.61, "p95_ms": 104.49, "p99_ms": 166.01} +{"operation": "batch_write", "items_per_request": 2, "clients": 1, "elapsed_s": 5.01, "completed": 157, "errors": 0, "first_error": null, "ops_per_s": 31.34, "items_per_s": 62.69, "mean_ms": 31.85, "p50_ms": 19.65, "p95_ms": 78.07, "p99_ms": 145.62} +{"operation": "batch_write", "items_per_request": 2, "clients": 8, "elapsed_s": 5.08, "completed": 164, "errors": 0, "first_error": null, "ops_per_s": 32.28, "items_per_s": 64.55, "mean_ms": 244.75, "p50_ms": 156.92, "p95_ms": 329.08, "p99_ms": 1988.06} +{"operation": "transact_write", "items_per_request": 2, "clients": 1, "elapsed_s": 5.01, "completed": 85, "errors": 0, "first_error": null, "ops_per_s": 16.97, "items_per_s": 33.93, "mean_ms": 58.93, "p50_ms": 27.15, "p95_ms": 126.79, "p99_ms": 276.1} +{"operation": "transact_write", "items_per_request": 2, "clients": 8, "elapsed_s": 5.15, "completed": 675, "errors": 0, "first_error": null, "ops_per_s": 131.07, "items_per_s": 262.15, "mean_ms": 60.13, "p50_ms": 41.7, "p95_ms": 170.65, "p99_ms": 282.78} diff --git a/benchmarks/2026-09-29-cellule-main-rerun/sqlite-meta.json b/benchmarks/2026-09-29-cellule-main-rerun/sqlite-meta.json new file mode 100644 index 0000000..15f74ba --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/sqlite-meta.json @@ -0,0 +1,24 @@ +{ + "server_commit": "7eaa89b437feed0af0f05883d3f1493f86c6fc6d", + "binary_sha256": "dfc9cd868d2bc06b71fa9fbc460a074dca330cb8ceffaebf62d8d54bdb485e38", + "host_load_start": [ + 31.0166015625, + 43.97607421875, + 51.90283203125 + ], + "logical_cpus": 12, + "boto3_version": "1.43.105", + "fixture": { + "sqlite_file": "/var/folders/rn/k2b7mvbs2dgb__6ys6_h94pm0000gn/T/extenddb-perf-rerun-tg7_lwhj/data.sqlite", + "dev_mode_open_auth": true + }, + "endpoint": "http://127.0.0.1:54301", + "started_at_unix": 1790751194.081158, + "host_load_end": [ + 37.271484375, + 41.0068359375, + 49.60595703125 + ], + "ended_at_unix": 1790751317.2063732, + "bench_exit_code": 0 +} diff --git a/benchmarks/2026-09-29-cellule-main-rerun/sqlite-run.json b/benchmarks/2026-09-29-cellule-main-rerun/sqlite-run.json new file mode 100644 index 0000000..b198e3a --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/sqlite-run.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T06:53:14.682071+00:00", + "endpoint": "http://127.0.0.1:54301", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1398, + "errors": 0, + "first_error": null, + "ops_per_s": 279.56, + "items_per_s": 279.56, + "mean_ms": 3.58, + "p50_ms": 2.42, + "p95_ms": 9.81, + "p99_ms": 18.8 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2541, + "errors": 0, + "first_error": null, + "ops_per_s": 507.24, + "items_per_s": 507.24, + "mean_ms": 15.74, + "p50_ms": 11.36, + "p95_ms": 30.47, + "p99_ms": 71.22 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1907, + "errors": 0, + "first_error": null, + "ops_per_s": 381.05, + "items_per_s": 381.05, + "mean_ms": 2.62, + "p50_ms": 1.86, + "p95_ms": 5.52, + "p99_ms": 13.14 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3057, + "errors": 0, + "first_error": null, + "ops_per_s": 610.54, + "items_per_s": 610.54, + "mean_ms": 13.08, + "p50_ms": 10.48, + "p95_ms": 29.65, + "p99_ms": 52.03 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1839, + "errors": 0, + "first_error": null, + "ops_per_s": 367.69, + "items_per_s": 367.69, + "mean_ms": 2.72, + "p50_ms": 2.1, + "p95_ms": 6.65, + "p99_ms": 11.03 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 3776, + "errors": 0, + "first_error": null, + "ops_per_s": 754.88, + "items_per_s": 754.88, + "mean_ms": 10.59, + "p50_ms": 8.56, + "p95_ms": 25.21, + "p99_ms": 38.19 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 936, + "errors": 0, + "first_error": null, + "ops_per_s": 187.15, + "items_per_s": 374.3, + "mean_ms": 5.34, + "p50_ms": 4.02, + "p95_ms": 12.96, + "p99_ms": 19.41 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2895, + "errors": 0, + "first_error": null, + "ops_per_s": 578.24, + "items_per_s": 1156.48, + "mean_ms": 13.82, + "p50_ms": 11.35, + "p95_ms": 28.67, + "p99_ms": 50.73 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.01, + "completed": 861, + "errors": 0, + "first_error": null, + "ops_per_s": 171.78, + "items_per_s": 343.57, + "mean_ms": 5.82, + "p50_ms": 3.3, + "p95_ms": 18.75, + "p99_ms": 37.54 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2370, + "errors": 0, + "first_error": null, + "ops_per_s": 473.26, + "items_per_s": 946.52, + "mean_ms": 16.87, + "p50_ms": 14.16, + "p95_ms": 33.01, + "p99_ms": 57.61 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.01, + "completed": 1296, + "errors": 0, + "first_error": null, + "ops_per_s": 258.86, + "items_per_s": 0.0, + "mean_ms": 3.86, + "p50_ms": 2.53, + "p95_ms": 10.4, + "p99_ms": 25.92 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2047, + "errors": 0, + "first_error": null, + "ops_per_s": 408.94, + "items_per_s": 0.0, + "mean_ms": 19.52, + "p50_ms": 15.54, + "p95_ms": 45.37, + "p99_ms": 81.94 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1796, + "errors": 0, + "first_error": null, + "ops_per_s": 359.12, + "items_per_s": 0.0, + "mean_ms": 2.78, + "p50_ms": 2.12, + "p95_ms": 5.47, + "p99_ms": 13.81 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.02, + "completed": 2493, + "errors": 0, + "first_error": null, + "ops_per_s": 496.83, + "items_per_s": 0.0, + "mean_ms": 16.05, + "p50_ms": 11.48, + "p95_ms": 35.28, + "p99_ms": 126.37 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.02, + "completed": 464, + "errors": 0, + "first_error": null, + "ops_per_s": 92.52, + "items_per_s": 92.52, + "mean_ms": 10.8, + "p50_ms": 4.84, + "p95_ms": 30.0, + "p99_ms": 64.34 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 274, + "errors": 0, + "first_error": null, + "ops_per_s": 54.71, + "items_per_s": 54.71, + "mean_ms": 145.99, + "p50_ms": 79.97, + "p95_ms": 311.05, + "p99_ms": 1633.93 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 286, + "errors": 0, + "first_error": null, + "ops_per_s": 57.18, + "items_per_s": 57.18, + "mean_ms": 17.49, + "p50_ms": 8.24, + "p95_ms": 58.88, + "p99_ms": 113.28 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.17, + "completed": 243, + "errors": 0, + "first_error": null, + "ops_per_s": 47.02, + "items_per_s": 47.02, + "mean_ms": 168.57, + "p50_ms": 140.05, + "p95_ms": 380.32, + "p99_ms": 762.03 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 427, + "errors": 0, + "first_error": null, + "ops_per_s": 85.32, + "items_per_s": 85.32, + "mean_ms": 11.71, + "p50_ms": 6.38, + "p95_ms": 38.59, + "p99_ms": 70.44 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.03, + "completed": 1242, + "errors": 0, + "first_error": null, + "ops_per_s": 246.95, + "items_per_s": 246.95, + "mean_ms": 32.29, + "p50_ms": 18.61, + "p95_ms": 104.49, + "p99_ms": 166.01 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.01, + "completed": 157, + "errors": 0, + "first_error": null, + "ops_per_s": 31.34, + "items_per_s": 62.69, + "mean_ms": 31.85, + "p50_ms": 19.65, + "p95_ms": 78.07, + "p99_ms": 145.62 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.08, + "completed": 164, + "errors": 0, + "first_error": null, + "ops_per_s": 32.28, + "items_per_s": 64.55, + "mean_ms": 244.75, + "p50_ms": 156.92, + "p95_ms": 329.08, + "p99_ms": 1988.06 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.01, + "completed": 85, + "errors": 0, + "first_error": null, + "ops_per_s": 16.97, + "items_per_s": 33.93, + "mean_ms": 58.93, + "p50_ms": 27.15, + "p95_ms": 126.79, + "p99_ms": 276.1 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.15, + "completed": 675, + "errors": 0, + "first_error": null, + "ops_per_s": 131.07, + "items_per_s": 262.15, + "mean_ms": 60.13, + "p50_ms": 41.7, + "p95_ms": 170.65, + "p99_ms": 282.78 + } + ] +} diff --git a/benchmarks/2026-09-29-cellule-main-rerun/sqlite-runner.log b/benchmarks/2026-09-29-cellule-main-rerun/sqlite-runner.log new file mode 100644 index 0000000..bfd9a62 --- /dev/null +++ b/benchmarks/2026-09-29-cellule-main-rerun/sqlite-runner.log @@ -0,0 +1,2 @@ +SQLite run started; host load: (31.0166015625, 43.97607421875, 51.90283203125) +SQLite run finished exit: 0 host load: (37.271484375, 41.0068359375, 49.60595703125) diff --git a/benchmarks/2026-09-29-claim-refresh/README.md b/benchmarks/2026-09-29-claim-refresh/README.md new file mode 100644 index 0000000..265b193 --- /dev/null +++ b/benchmarks/2026-09-29-claim-refresh/README.md @@ -0,0 +1,67 @@ +# September 29, 2026 performance claim refresh + +These are raw results from two independent local RustFS fixtures. They test +whether the earlier rates quoted in PR #14 and `docs/performance.md` reproduce +on the current `af8fab7` release binary. We could not locate raw output from +the earlier `880b4aa` sample, so this is a refresh, not a replay of its +exact host state or code revision. + +## Fixture + +- Mac14,13: 12 logical CPUs, 32 GiB RAM. Other virtual machines and Rust + builds were active. The one-minute host load average was 16.8 at the start + of run 1 and 35.1 at the start of run 2, and rose above 30 during testing. +- A new RustFS container and bucket for each run, pinned to image + `ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. + Containers were stopped after the tests. +- BeyondDB `af8fab7`, `cargo build --locked --release --bin beyonddb`, four + initial partitions, `auth_cache_enabled: true`, 12 SQL workers, 128 active + Cell slots, and a 1 GiB node retained-byte budget. +- One PAY_PER_REQUEST table with a string `pk`, no indexes or Streams, 64 + seeded items with a 1 KiB `payload` attribute, signed boto3 requests, and + zero SDK retries. +- `scripts/bench.py` used five seconds per API and client count, testing one + then eight clients. Batch and transaction calls contained two items. + +## Outcome + +| API | Run 2 requests/s, 1 / 8 clients | Run 2 p95 ms, 1 / 8 clients | +| --- | ---: | ---: | +| GetItem | 164.13 / 549.53 | 15.62 / 35.71 | +| Query | 164.70 / 475.40 | 16.67 / 43.78 | +| Scan | 119.55 / 413.38 | 21.44 / 52.36 | +| BatchGetItem | 85.75 / 323.10 | 40.10 / 54.50 | +| TransactGetItems | 0.48 / 2.32 | 2,294.76 / 4,399.31 | +| DescribeTable | 193.18 / 327.33 | 10.84 / 52.39 | +| ListTables | 202.16 / 624.79 | 7.71 / 26.95 | +| PutItem | 10.13 / 31.56 | 211.23 / 682.53 | +| UpdateItem | 7.51 / 11.95 | 201.03 / 1,518.74 | +| DeleteItem, absent key | 6.68 / 35.03 | 302.84 / 444.75 | +| BatchWriteItem | 4.66 / 20.98 | 364.55 / 858.28 | +| TransactWriteItems | 0.74 / 2.10 | 1,368.98 / 4,174.11 | + +Run 2 completed all 24 cases with zero SDK request errors. Run 1 stopped at +eight-client `TransactGetItems` after eight SDK read timeouts; only its first +ten cases were recorded. Both server logs recorded three deferred background +operations caused by Cell mailbox-byte exhaustion. The completed run therefore +does not establish sustainable capacity. The high host load and code changes +since `880b4aa` prevent attribution of the lower rates to one cause. + +See [run 1](run-1.json), [run 2](run-2.json), +[fixture metadata](summary.json), and the server warnings from +[run 1](warnings-run-1.txt) and [run 2](warnings-run-2.txt). The benchmark +invocation, after starting a fresh fixture and setting AWS credentials for the +server's bootstrap identity, was: + +```sh +python3 scripts/bench.py \ + --endpoint http://127.0.0.1:8000 --table PerfData \ + --seconds 5 --clients 1 8 --payload-bytes 1024 \ + --operations get query scan batch_get transact_get describe_table list_tables \ + put update delete_missing batch_write transact_write \ + --output results.json +``` + +The `8000` endpoint is illustrative; the recorded runs used ephemeral local +ports. See [deployment configuration](../../docs/deployment.md) for fixture +setup. diff --git a/benchmarks/2026-09-29-claim-refresh/run-1.json b/benchmarks/2026-09-29-claim-refresh/run-1.json new file mode 100644 index 0000000..d5254fb --- /dev/null +++ b/benchmarks/2026-09-29-claim-refresh/run-1.json @@ -0,0 +1,177 @@ +{ + "started_at": "2026-09-29T23:47:35.319047+00:00", + "endpoint": "http://127.0.0.1:59416", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1830, + "errors": 0, + "first_error": null, + "ops_per_s": 365.88, + "items_per_s": 365.88, + "mean_ms": 2.73, + "p50_ms": 2.08, + "p95_ms": 5.89, + "p99_ms": 9.67 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2972, + "errors": 0, + "first_error": null, + "ops_per_s": 593.65, + "items_per_s": 593.65, + "mean_ms": 13.45, + "p50_ms": 11.12, + "p95_ms": 28.64, + "p99_ms": 49.14 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1061, + "errors": 0, + "first_error": null, + "ops_per_s": 212.1, + "items_per_s": 212.1, + "mean_ms": 4.71, + "p50_ms": 3.39, + "p95_ms": 11.74, + "p99_ms": 22.77 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 2891, + "errors": 0, + "first_error": null, + "ops_per_s": 577.66, + "items_per_s": 577.66, + "mean_ms": 13.83, + "p50_ms": 11.31, + "p95_ms": 31.12, + "p99_ms": 59.45 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 1093, + "errors": 0, + "first_error": null, + "ops_per_s": 218.33, + "items_per_s": 218.33, + "mean_ms": 4.58, + "p50_ms": 3.21, + "p95_ms": 10.2, + "p99_ms": 16.7 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.02, + "completed": 2517, + "errors": 0, + "first_error": null, + "ops_per_s": 501.64, + "items_per_s": 501.64, + "mean_ms": 15.91, + "p50_ms": 12.03, + "p95_ms": 37.42, + "p99_ms": 87.81 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.03, + "completed": 677, + "errors": 0, + "first_error": null, + "ops_per_s": 134.63, + "items_per_s": 269.26, + "mean_ms": 7.43, + "p50_ms": 4.26, + "p95_ms": 21.3, + "p99_ms": 62.05 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.01, + "completed": 1850, + "errors": 0, + "first_error": null, + "ops_per_s": 369.01, + "items_per_s": 738.02, + "mean_ms": 21.63, + "p50_ms": 16.42, + "p95_ms": 53.72, + "p99_ms": 79.9 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.19, + "completed": 4, + "errors": 0, + "first_error": null, + "ops_per_s": 0.77, + "items_per_s": 1.54, + "mean_ms": 1295.76, + "p50_ms": 1036.98, + "p95_ms": 1908.44, + "p99_ms": 1908.44 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 10.03, + "completed": 1, + "errors": 8, + "first_error": "ReadTimeoutError: Read timeout on endpoint URL: \"http://127.0.0.1:59416/\"", + "ops_per_s": 0.1, + "items_per_s": 0.2, + "mean_ms": 13.25, + "p50_ms": 13.25, + "p95_ms": 13.25, + "p99_ms": 13.25 + } + ] +} diff --git a/benchmarks/2026-09-29-claim-refresh/run-2.json b/benchmarks/2026-09-29-claim-refresh/run-2.json new file mode 100644 index 0000000..a52a2e1 --- /dev/null +++ b/benchmarks/2026-09-29-claim-refresh/run-2.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-29T23:48:57.173218+00:00", + "endpoint": "http://127.0.0.1:52346", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 821, + "errors": 0, + "first_error": null, + "ops_per_s": 164.13, + "items_per_s": 164.13, + "mean_ms": 6.09, + "p50_ms": 3.47, + "p95_ms": 15.62, + "p99_ms": 35.56 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2753, + "errors": 0, + "first_error": null, + "ops_per_s": 549.53, + "items_per_s": 549.53, + "mean_ms": 14.54, + "p50_ms": 11.56, + "p95_ms": 35.71, + "p99_ms": 60.78 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 824, + "errors": 0, + "first_error": null, + "ops_per_s": 164.7, + "items_per_s": 164.7, + "mean_ms": 6.07, + "p50_ms": 3.91, + "p95_ms": 16.67, + "p99_ms": 34.47 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.02, + "completed": 2386, + "errors": 0, + "first_error": null, + "ops_per_s": 475.4, + "items_per_s": 475.4, + "mean_ms": 16.77, + "p50_ms": 13.23, + "p95_ms": 43.78, + "p99_ms": 71.13 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.05, + "completed": 604, + "errors": 0, + "first_error": null, + "ops_per_s": 119.55, + "items_per_s": 119.55, + "mean_ms": 8.36, + "p50_ms": 5.35, + "p95_ms": 21.44, + "p99_ms": 49.41 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.15, + "completed": 2127, + "errors": 0, + "first_error": null, + "ops_per_s": 413.38, + "items_per_s": 413.38, + "mean_ms": 18.95, + "p50_ms": 13.1, + "p95_ms": 52.36, + "p99_ms": 99.55 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 429, + "errors": 0, + "first_error": null, + "ops_per_s": 85.75, + "items_per_s": 171.49, + "mean_ms": 11.66, + "p50_ms": 5.95, + "p95_ms": 40.1, + "p99_ms": 84.81 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.01, + "completed": 1619, + "errors": 0, + "first_error": null, + "ops_per_s": 323.1, + "items_per_s": 646.2, + "mean_ms": 24.7, + "p50_ms": 17.68, + "p95_ms": 54.5, + "p99_ms": 199.06 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 6.3, + "completed": 3, + "errors": 0, + "first_error": null, + "ops_per_s": 0.48, + "items_per_s": 0.95, + "mean_ms": 2101.32, + "p50_ms": 2294.76, + "p95_ms": 2294.76, + "p99_ms": 2294.76 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 7.33, + "completed": 17, + "errors": 0, + "first_error": null, + "ops_per_s": 2.32, + "items_per_s": 4.64, + "mean_ms": 2948.37, + "p50_ms": 2919.58, + "p95_ms": 4399.31, + "p99_ms": 4399.31 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 966, + "errors": 0, + "first_error": null, + "ops_per_s": 193.18, + "items_per_s": 0.0, + "mean_ms": 5.17, + "p50_ms": 3.45, + "p95_ms": 10.84, + "p99_ms": 50.89 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 1641, + "errors": 0, + "first_error": null, + "ops_per_s": 327.33, + "items_per_s": 0.0, + "mean_ms": 24.4, + "p50_ms": 17.51, + "p95_ms": 52.39, + "p99_ms": 105.84 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1011, + "errors": 0, + "first_error": null, + "ops_per_s": 202.16, + "items_per_s": 0.0, + "mean_ms": 4.94, + "p50_ms": 2.57, + "p95_ms": 7.71, + "p99_ms": 12.61 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.0, + "completed": 3126, + "errors": 0, + "first_error": null, + "ops_per_s": 624.79, + "items_per_s": 0.0, + "mean_ms": 12.79, + "p50_ms": 10.79, + "p95_ms": 26.95, + "p99_ms": 54.09 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.13, + "completed": 52, + "errors": 0, + "first_error": null, + "ops_per_s": 10.13, + "items_per_s": 10.13, + "mean_ms": 98.67, + "p50_ms": 72.26, + "p95_ms": 211.23, + "p99_ms": 313.77 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.42, + "completed": 171, + "errors": 0, + "first_error": null, + "ops_per_s": 31.56, + "items_per_s": 31.56, + "mean_ms": 247.16, + "p50_ms": 181.11, + "p95_ms": 682.53, + "p99_ms": 942.69 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.06, + "completed": 38, + "errors": 0, + "first_error": null, + "ops_per_s": 7.51, + "items_per_s": 7.51, + "mean_ms": 133.21, + "p50_ms": 117.66, + "p95_ms": 201.03, + "p99_ms": 224.4 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.77, + "completed": 69, + "errors": 0, + "first_error": null, + "ops_per_s": 11.95, + "items_per_s": 11.95, + "mean_ms": 612.3, + "p50_ms": 449.96, + "p95_ms": 1518.74, + "p99_ms": 2226.36 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.09, + "completed": 34, + "errors": 0, + "first_error": null, + "ops_per_s": 6.68, + "items_per_s": 6.68, + "mean_ms": 149.68, + "p50_ms": 129.04, + "p95_ms": 302.84, + "p99_ms": 421.96 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.2, + "completed": 182, + "errors": 0, + "first_error": null, + "ops_per_s": 35.03, + "items_per_s": 35.03, + "mean_ms": 222.45, + "p50_ms": 182.45, + "p95_ms": 444.75, + "p99_ms": 862.75 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.15, + "completed": 24, + "errors": 0, + "first_error": null, + "ops_per_s": 4.66, + "items_per_s": 9.32, + "mean_ms": 214.57, + "p50_ms": 196.91, + "p95_ms": 364.55, + "p99_ms": 384.67 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.24, + "completed": 110, + "errors": 0, + "first_error": null, + "ops_per_s": 20.98, + "items_per_s": 41.97, + "mean_ms": 374.39, + "p50_ms": 255.12, + "p95_ms": 858.28, + "p99_ms": 938.18 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.43, + "completed": 4, + "errors": 0, + "first_error": null, + "ops_per_s": 0.74, + "items_per_s": 1.47, + "mean_ms": 1356.7, + "p50_ms": 1342.67, + "p95_ms": 1368.98, + "p99_ms": 1368.98 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 8.11, + "completed": 17, + "errors": 0, + "first_error": null, + "ops_per_s": 2.1, + "items_per_s": 4.19, + "mean_ms": 3211.68, + "p50_ms": 3246.99, + "p95_ms": 4174.11, + "p99_ms": 4174.11 + } + ] +} diff --git a/benchmarks/2026-09-29-claim-refresh/summary.json b/benchmarks/2026-09-29-claim-refresh/summary.json new file mode 100644 index 0000000..e9cdc2c --- /dev/null +++ b/benchmarks/2026-09-29-claim-refresh/summary.json @@ -0,0 +1,60 @@ +[ + { + "run": 1, + "git_revision": "af8fab7", + "release_binary": "/Users/haipingfu/Workspace/crabbuild-target/beyonddb-duplicate-batcher-check-20260929/release/beyonddb", + "payload_bytes": 1024, + "seed_keys": 64, + "initial_partitions": 4, + "auth_cache_enabled": true, + "sql_workers": 12, + "node_retained_bytes": 1073741824, + "seconds_per_case": 5, + "clients": [ + 1, + 8 + ], + "sdk_retries": 0, + "benchmark_exit_code": 1, + "host_load_end": [ + 30.6328125, + 24.6064453125, + 23.73291015625 + ], + "host_load_start": [ + 16.84033203125, + 21.82666015625, + 22.78173828125 + ], + "server_warning_count": 3 + }, + { + "run": 2, + "git_revision": "af8fab7", + "release_binary": "/Users/haipingfu/Workspace/crabbuild-target/beyonddb-duplicate-batcher-check-20260929/release/beyonddb", + "payload_bytes": 1024, + "seed_keys": 64, + "initial_partitions": 4, + "auth_cache_enabled": true, + "sql_workers": 12, + "node_retained_bytes": 1073741824, + "seconds_per_case": 5, + "clients": [ + 1, + 8 + ], + "sdk_retries": 0, + "benchmark_exit_code": 0, + "host_load_end": [ + 34.5322265625, + 28.69775390625, + 25.5830078125 + ], + "host_load_start": [ + 35.06640625, + 25.62548828125, + 24.09765625 + ], + "server_warning_count": 3 + } +] diff --git a/benchmarks/2026-09-29-claim-refresh/warnings-run-1.txt b/benchmarks/2026-09-29-claim-refresh/warnings-run-1.txt new file mode 100644 index 0000000..5253566 --- /dev/null +++ b/benchmarks/2026-09-29-claim-refresh/warnings-run-1.txt @@ -0,0 +1,3 @@ +2026-09-29T23:48:11.622384Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-29T23:48:13.727582Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-29T23:48:14.227663Z WARN beyonddb::provision::capacity: capacity sweep deferred account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes diff --git a/benchmarks/2026-09-29-claim-refresh/warnings-run-2.txt b/benchmarks/2026-09-29-claim-refresh/warnings-run-2.txt new file mode 100644 index 0000000..354fd4c --- /dev/null +++ b/benchmarks/2026-09-29-claim-refresh/warnings-run-2.txt @@ -0,0 +1,3 @@ +2026-09-29T23:49:36.304803Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-29T23:49:57.104437Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-29T23:49:59.204136Z WARN beyonddb::provision::capacity: capacity sweep deferred account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes diff --git a/benchmarks/2026-09-29-current-release-rerun/README.md b/benchmarks/2026-09-29-current-release-rerun/README.md new file mode 100644 index 0000000..807e38d --- /dev/null +++ b/benchmarks/2026-09-29-current-release-rerun/README.md @@ -0,0 +1,65 @@ +# Current release rerun against ExtendDB SQLite + +On September 29, 2026 (Pacific time), the BeyondDB release binary from +`7655c608cb7020836459502010aed787fc48bb1c` and the pinned ExtendDB +SQLite release from `7eaa89b437feed0af0f05883d3f1493f86c6fc6d` ran the +same signed `scripts/bench.py` workload, sequentially, on one Mac14,13 host. +The BeyondDB binary was rebuilt with `cargo build --release --locked --bin +beyonddb`; its SHA-256 is +`01c4842a804689f7705e3388f0450d3a3778ab2d259de9977abc2afdead536af`. +The ExtendDB binary reports version `0.1.12`, catalog `0.0.3 (sqlite)`, and +commit `7eaa89b`; its SHA-256 is +`dfc9cd868d2bc06b71fa9fbc460a074dca330cb8ceffaebf62d8d54bdb485e38`. + +Both runs used fresh tables, 64 seeded items with 1 KiB payloads, boto3 +`1.43.105`, zero SDK retries, one and eight closed-loop clients, and five +seconds per case. Batch and transaction requests carried two items. `GetItem`, +`Query`, `Scan`, and `BatchGetItem` requested consistent reads. `DeleteItem` +targeted a unique absent key. Each backend completed all 24 cases with zero +foreground SDK errors. + +| API | BeyondDB requests/s, 1 / 8 clients | SQLite requests/s, 1 / 8 clients | BeyondDB p95 ms, 1 / 8 clients | SQLite p95 ms, 1 / 8 clients | +| --- | ---: | ---: | ---: | ---: | +| GetItem | 346.56 / 723.43 | 545.85 / 898.90 | 7.17 / 23.39 | 3.63 / 17.83 | +| Query | 385.50 / 748.14 | 504.04 / 764.27 | 6.16 / 22.84 | 3.60 / 21.40 | +| Scan | 322.35 / 654.17 | 454.04 / 791.91 | 5.82 / 23.02 | 5.08 / 19.97 | +| BatchGetItem | 149.32 / 390.22 | 432.05 / 784.41 | 17.46 / 46.55 | 4.87 / 19.94 | +| TransactGetItems | 1.34 / 5.62 | 577.47 / 1,108.67 | 1,021.77 / 1,894.27 | 3.24 / 14.07 | +| DescribeTable | 518.99 / 684.89 | 875.52 / 1,126.43 | 3.30 / 28.58 | 2.08 / 14.11 | +| ListTables | 584.98 / 896.89 | 783.11 / 939.65 | 3.29 / 19.73 | 2.59 / 18.17 | +| PutItem | 24.29 / 82.97 | 377.16 / 358.10 | 67.36 / 223.29 | 6.81 / 60.32 | +| UpdateItem | 21.00 / 51.32 | 104.34 / 268.37 | 94.74 / 344.00 | 25.88 / 79.00 | +| DeleteItem, absent key | 15.22 / 46.14 | 150.10 / 527.73 | 125.13 / 359.98 | 19.06 / 35.27 | +| BatchWriteItem | 6.20 / 27.95 | 122.42 / 221.15 | 322.46 / 518.20 | 23.70 / 96.72 | +| TransactWriteItems | 1.24 / 2.94 | 37.11 / 145.89 | 904.19 / 3,480.67 | 72.41 / 101.73 | + +`BatchGetItem` completed 298.63/780.44 items/s in BeyondDB and +864.10/1,568.82 items/s in SQLite. `BatchWriteItem` completed 12.39/55.91 +items/s in BeyondDB and 244.84/442.30 items/s in SQLite. ExtendDB SQLite +was faster in every API and client-count case in this rerun. The large +transaction gap and durable write gap remain open. + +## Fixture and limits + +- BeyondDB used four initial partitions, twelve SQL workers, 128 active Cell + slots, `auth_cache_enabled: true`, a 1 GiB opt-in persistent follower-store + budget, and a fresh RustFS bucket. RustFS used the pinned image + `ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. + The follower receiver did **not** enable follower-backed commit proofs; the + serving path still awaited object-store publication. BeyondDB verified IAM. +- ExtendDB used its file-backed SQLite `sqlite,dev-mode` release. Dev mode + verified SigV4 and used open authorization. Its local-file durability and + authorization contract therefore differed from BeyondDB's. +- The twelve-logical-CPU host was contended. The one-minute load average rose + from 15.07 to 22.64 during BeyondDB and from 18.75 to 23.40 during SQLite. + Other virtual machines were active, so the sequential numbers cannot prove + a stable causal speed ratio or production capacity. +- BeyondDB logged four deferred background operations: three global-index + projection deferrals and one capacity-sweep deferral, all reporting Cell + mailbox-byte exhaustion. Zero SDK errors do not establish sustainable + throughput while this background work falls behind. + +See [BeyondDB raw cases](run.json), [SQLite raw cases](sqlite-run.json), +[BeyondDB fixture metadata](meta.json), [SQLite fixture metadata](sqlite-meta.json), +and the [BeyondDB server warnings](server-warnings.txt). The JSON contains +counts, elapsed time, throughput, item rates, and p50/p95/p99 for every case. diff --git a/benchmarks/2026-09-29-current-release-rerun/meta.json b/benchmarks/2026-09-29-current-release-rerun/meta.json new file mode 100644 index 0000000..d62c36c --- /dev/null +++ b/benchmarks/2026-09-29-current-release-rerun/meta.json @@ -0,0 +1,29 @@ +{ + "server_commit": "7655c608cb7020836459502010aed787fc48bb1c", + "binary_sha256": "01c4842a804689f7705e3388f0450d3a3778ab2d259de9977abc2afdead536af", + "rustfs_image": "ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858", + "host_load_start": [ + 15.072265625, + 19.65478515625, + 21.03076171875 + ], + "logical_cpus": 12, + "boto3_version": "1.43.105", + "fixture": { + "initial_partitions": 4, + "sql_workers": 12, + "max_active_cells": 128, + "follower_store_bytes": 1073741824, + "auth_cache_enabled": true, + "server_object_publication": true + }, + "endpoint": "http://127.0.0.1:62102", + "started_at_unix": 1790737358.9334168, + "host_load_end": [ + 22.64404296875, + 21.43017578125, + 21.5693359375 + ], + "ended_at_unix": 1790737488.844814, + "bench_exit_code": 0 +} diff --git a/benchmarks/2026-09-29-current-release-rerun/run.json b/benchmarks/2026-09-29-current-release-rerun/run.json new file mode 100644 index 0000000..69e49b1 --- /dev/null +++ b/benchmarks/2026-09-29-current-release-rerun/run.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T03:02:43.791356+00:00", + "endpoint": "http://127.0.0.1:62102", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1733, + "errors": 0, + "first_error": null, + "ops_per_s": 346.56, + "items_per_s": 346.56, + "mean_ms": 2.88, + "p50_ms": 2.01, + "p95_ms": 7.17, + "p99_ms": 14.39 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3621, + "errors": 0, + "first_error": null, + "ops_per_s": 723.43, + "items_per_s": 723.43, + "mean_ms": 11.05, + "p50_ms": 9.45, + "p95_ms": 23.39, + "p99_ms": 35.41 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1928, + "errors": 0, + "first_error": null, + "ops_per_s": 385.5, + "items_per_s": 385.5, + "mean_ms": 2.59, + "p50_ms": 2.07, + "p95_ms": 6.16, + "p99_ms": 10.03 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3748, + "errors": 0, + "first_error": null, + "ops_per_s": 748.14, + "items_per_s": 748.14, + "mean_ms": 10.68, + "p50_ms": 9.16, + "p95_ms": 22.84, + "p99_ms": 36.19 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1612, + "errors": 0, + "first_error": null, + "ops_per_s": 322.35, + "items_per_s": 322.35, + "mean_ms": 3.1, + "p50_ms": 2.46, + "p95_ms": 5.82, + "p99_ms": 11.78 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3275, + "errors": 0, + "first_error": null, + "ops_per_s": 654.17, + "items_per_s": 654.17, + "mean_ms": 12.22, + "p50_ms": 10.84, + "p95_ms": 23.02, + "p99_ms": 32.2 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.01, + "completed": 748, + "errors": 0, + "first_error": null, + "ops_per_s": 149.32, + "items_per_s": 298.63, + "mean_ms": 6.7, + "p50_ms": 4.5, + "p95_ms": 17.46, + "p99_ms": 36.57 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.03, + "completed": 1963, + "errors": 0, + "first_error": null, + "ops_per_s": 390.22, + "items_per_s": 780.44, + "mean_ms": 20.44, + "p50_ms": 16.06, + "p95_ms": 46.55, + "p99_ms": 74.87 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.23, + "completed": 7, + "errors": 0, + "first_error": null, + "ops_per_s": 1.34, + "items_per_s": 2.68, + "mean_ms": 747.2, + "p50_ms": 702.61, + "p95_ms": 1021.77, + "p99_ms": 1021.77 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 6.23, + "completed": 35, + "errors": 0, + "first_error": null, + "ops_per_s": 5.62, + "items_per_s": 11.24, + "mean_ms": 1350.32, + "p50_ms": 1539.76, + "p95_ms": 1894.27, + "p99_ms": 2002.76 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2595, + "errors": 0, + "first_error": null, + "ops_per_s": 518.99, + "items_per_s": 0.0, + "mean_ms": 1.92, + "p50_ms": 1.65, + "p95_ms": 3.3, + "p99_ms": 6.58 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3428, + "errors": 0, + "first_error": null, + "ops_per_s": 684.89, + "items_per_s": 0.0, + "mean_ms": 11.67, + "p50_ms": 9.04, + "p95_ms": 28.58, + "p99_ms": 52.06 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2925, + "errors": 0, + "first_error": null, + "ops_per_s": 584.98, + "items_per_s": 0.0, + "mean_ms": 1.71, + "p50_ms": 1.49, + "p95_ms": 3.29, + "p99_ms": 5.57 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.0, + "completed": 4487, + "errors": 0, + "first_error": null, + "ops_per_s": 896.89, + "items_per_s": 0.0, + "mean_ms": 8.91, + "p50_ms": 7.52, + "p95_ms": 19.73, + "p99_ms": 27.76 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.02, + "completed": 122, + "errors": 0, + "first_error": null, + "ops_per_s": 24.29, + "items_per_s": 24.29, + "mean_ms": 41.16, + "p50_ms": 37.29, + "p95_ms": 67.36, + "p99_ms": 87.84 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.18, + "completed": 430, + "errors": 0, + "first_error": null, + "ops_per_s": 82.97, + "items_per_s": 82.97, + "mean_ms": 93.8, + "p50_ms": 78.46, + "p95_ms": 223.29, + "p99_ms": 265.86 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 105, + "errors": 0, + "first_error": null, + "ops_per_s": 21.0, + "items_per_s": 21.0, + "mean_ms": 47.62, + "p50_ms": 35.88, + "p95_ms": 94.74, + "p99_ms": 130.45 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.14, + "completed": 264, + "errors": 0, + "first_error": null, + "ops_per_s": 51.32, + "items_per_s": 51.32, + "mean_ms": 154.24, + "p50_ms": 131.04, + "p95_ms": 344.0, + "p99_ms": 438.65 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.12, + "completed": 78, + "errors": 0, + "first_error": null, + "ops_per_s": 15.22, + "items_per_s": 15.22, + "mean_ms": 65.64, + "p50_ms": 56.03, + "p95_ms": 125.13, + "p99_ms": 149.64 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.27, + "completed": 243, + "errors": 0, + "first_error": null, + "ops_per_s": 46.14, + "items_per_s": 46.14, + "mean_ms": 166.73, + "p50_ms": 133.41, + "p95_ms": 359.98, + "p99_ms": 595.01 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 31, + "errors": 0, + "first_error": null, + "ops_per_s": 6.2, + "items_per_s": 12.39, + "mean_ms": 161.36, + "p50_ms": 157.46, + "p95_ms": 322.46, + "p99_ms": 354.99 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.15, + "completed": 144, + "errors": 0, + "first_error": null, + "ops_per_s": 27.95, + "items_per_s": 55.91, + "mean_ms": 281.84, + "p50_ms": 237.05, + "p95_ms": 518.2, + "p99_ms": 682.05 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.63, + "completed": 7, + "errors": 0, + "first_error": null, + "ops_per_s": 1.24, + "items_per_s": 2.49, + "mean_ms": 804.0, + "p50_ms": 715.48, + "p95_ms": 904.19, + "p99_ms": 904.19 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 6.79, + "completed": 20, + "errors": 0, + "first_error": null, + "ops_per_s": 2.94, + "items_per_s": 5.89, + "mean_ms": 2488.13, + "p50_ms": 2310.15, + "p95_ms": 3480.67, + "p99_ms": 3480.67 + } + ] +} diff --git a/benchmarks/2026-09-29-current-release-rerun/server-warnings.txt b/benchmarks/2026-09-29-current-release-rerun/server-warnings.txt new file mode 100644 index 0000000..1d05e18 --- /dev/null +++ b/benchmarks/2026-09-29-current-release-rerun/server-warnings.txt @@ -0,0 +1,4 @@ +2026-09-30T03:03:19.778696Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T03:03:21.978125Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T03:03:29.167725Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T03:04:46.765835Z WARN beyonddb::provision::capacity: capacity sweep deferred account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes diff --git a/benchmarks/2026-09-29-current-release-rerun/sqlite-meta.json b/benchmarks/2026-09-29-current-release-rerun/sqlite-meta.json new file mode 100644 index 0000000..c8b73ae --- /dev/null +++ b/benchmarks/2026-09-29-current-release-rerun/sqlite-meta.json @@ -0,0 +1,24 @@ +{ + "server_commit": "7eaa89b437feed0af0f05883d3f1493f86c6fc6d", + "binary_sha256": "dfc9cd868d2bc06b71fa9fbc460a074dca330cb8ceffaebf62d8d54bdb485e38", + "host_load_start": [ + 18.75244140625, + 20.5693359375, + 21.23828125 + ], + "logical_cpus": 12, + "boto3_version": "1.43.105", + "fixture": { + "sqlite_file": "/var/folders/rn/k2b7mvbs2dgb__6ys6_h94pm0000gn/T/extenddb-perf-current-kgusnjz2/data.sqlite", + "dev_mode_open_auth": true + }, + "endpoint": "http://127.0.0.1:63533", + "started_at_unix": 1790737549.836853, + "host_load_end": [ + 23.40478515625, + 21.1689453125, + 21.34765625 + ], + "ended_at_unix": 1790737671.039475, + "bench_exit_code": 0 +} diff --git a/benchmarks/2026-09-29-current-release-rerun/sqlite-run.json b/benchmarks/2026-09-29-current-release-rerun/sqlite-run.json new file mode 100644 index 0000000..732b35c --- /dev/null +++ b/benchmarks/2026-09-29-current-release-rerun/sqlite-run.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T03:05:50.490261+00:00", + "endpoint": "http://127.0.0.1:63533", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2730, + "errors": 0, + "first_error": null, + "ops_per_s": 545.85, + "items_per_s": 545.85, + "mean_ms": 1.83, + "p50_ms": 1.51, + "p95_ms": 3.63, + "p99_ms": 6.71 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 4499, + "errors": 0, + "first_error": null, + "ops_per_s": 898.9, + "items_per_s": 898.9, + "mean_ms": 8.89, + "p50_ms": 7.97, + "p95_ms": 17.83, + "p99_ms": 23.11 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2521, + "errors": 0, + "first_error": null, + "ops_per_s": 504.04, + "items_per_s": 504.04, + "mean_ms": 1.98, + "p50_ms": 1.78, + "p95_ms": 3.6, + "p99_ms": 6.18 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 3825, + "errors": 0, + "first_error": null, + "ops_per_s": 764.27, + "items_per_s": 764.27, + "mean_ms": 10.45, + "p50_ms": 9.39, + "p95_ms": 21.4, + "p99_ms": 27.91 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2271, + "errors": 0, + "first_error": null, + "ops_per_s": 454.04, + "items_per_s": 454.04, + "mean_ms": 2.2, + "p50_ms": 1.77, + "p95_ms": 5.08, + "p99_ms": 8.85 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3965, + "errors": 0, + "first_error": null, + "ops_per_s": 791.91, + "items_per_s": 791.91, + "mean_ms": 10.09, + "p50_ms": 9.14, + "p95_ms": 19.97, + "p99_ms": 26.92 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2161, + "errors": 0, + "first_error": null, + "ops_per_s": 432.05, + "items_per_s": 864.1, + "mean_ms": 2.31, + "p50_ms": 1.9, + "p95_ms": 4.87, + "p99_ms": 8.71 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.0, + "completed": 3925, + "errors": 0, + "first_error": null, + "ops_per_s": 784.41, + "items_per_s": 1568.82, + "mean_ms": 10.19, + "p50_ms": 9.24, + "p95_ms": 19.94, + "p99_ms": 27.05 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2888, + "errors": 0, + "first_error": null, + "ops_per_s": 577.47, + "items_per_s": 1154.93, + "mean_ms": 1.73, + "p50_ms": 1.43, + "p95_ms": 3.24, + "p99_ms": 7.73 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.01, + "completed": 5557, + "errors": 0, + "first_error": null, + "ops_per_s": 1108.67, + "items_per_s": 2217.34, + "mean_ms": 7.21, + "p50_ms": 6.45, + "p95_ms": 14.07, + "p99_ms": 19.53 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 4379, + "errors": 0, + "first_error": null, + "ops_per_s": 875.52, + "items_per_s": 0.0, + "mean_ms": 1.14, + "p50_ms": 0.98, + "p95_ms": 2.08, + "p99_ms": 3.22 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.0, + "completed": 5635, + "errors": 0, + "first_error": null, + "ops_per_s": 1126.43, + "items_per_s": 0.0, + "mean_ms": 7.1, + "p50_ms": 6.25, + "p95_ms": 14.11, + "p99_ms": 20.79 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 3916, + "errors": 0, + "first_error": null, + "ops_per_s": 783.11, + "items_per_s": 0.0, + "mean_ms": 1.28, + "p50_ms": 1.07, + "p95_ms": 2.59, + "p99_ms": 5.03 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.0, + "completed": 4701, + "errors": 0, + "first_error": null, + "ops_per_s": 939.65, + "items_per_s": 0.0, + "mean_ms": 8.51, + "p50_ms": 7.35, + "p95_ms": 18.17, + "p99_ms": 27.14 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1886, + "errors": 0, + "first_error": null, + "ops_per_s": 377.16, + "items_per_s": 377.16, + "mean_ms": 2.65, + "p50_ms": 1.94, + "p95_ms": 6.81, + "p99_ms": 13.27 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.06, + "completed": 1813, + "errors": 0, + "first_error": null, + "ops_per_s": 358.1, + "items_per_s": 358.1, + "mean_ms": 22.2, + "p50_ms": 16.57, + "p95_ms": 60.32, + "p99_ms": 92.11 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 522, + "errors": 0, + "first_error": null, + "ops_per_s": 104.34, + "items_per_s": 104.34, + "mean_ms": 9.58, + "p50_ms": 6.3, + "p95_ms": 25.88, + "p99_ms": 40.18 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 1345, + "errors": 0, + "first_error": null, + "ops_per_s": 268.37, + "items_per_s": 268.37, + "mean_ms": 29.76, + "p50_ms": 21.96, + "p95_ms": 79.0, + "p99_ms": 105.65 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 751, + "errors": 0, + "first_error": null, + "ops_per_s": 150.1, + "items_per_s": 150.1, + "mean_ms": 6.66, + "p50_ms": 4.06, + "p95_ms": 19.06, + "p99_ms": 39.76 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 2640, + "errors": 0, + "first_error": null, + "ops_per_s": 527.73, + "items_per_s": 527.73, + "mean_ms": 15.15, + "p50_ms": 12.19, + "p95_ms": 35.27, + "p99_ms": 49.75 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.01, + "completed": 613, + "errors": 0, + "first_error": null, + "ops_per_s": 122.42, + "items_per_s": 244.84, + "mean_ms": 8.17, + "p50_ms": 5.42, + "p95_ms": 23.7, + "p99_ms": 38.27 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.15, + "completed": 1138, + "errors": 0, + "first_error": null, + "ops_per_s": 221.15, + "items_per_s": 442.3, + "mean_ms": 35.69, + "p50_ms": 22.5, + "p95_ms": 96.72, + "p99_ms": 220.15 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.04, + "completed": 187, + "errors": 0, + "first_error": null, + "ops_per_s": 37.11, + "items_per_s": 74.23, + "mean_ms": 26.93, + "p50_ms": 18.96, + "p95_ms": 72.41, + "p99_ms": 102.38 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.03, + "completed": 734, + "errors": 0, + "first_error": null, + "ops_per_s": 145.89, + "items_per_s": 291.79, + "mean_ms": 54.63, + "p50_ms": 49.67, + "p95_ms": 101.73, + "p99_ms": 140.3 + } + ] +} diff --git a/benchmarks/2026-09-29-follower-receiver-check/README.md b/benchmarks/2026-09-29-follower-receiver-check/README.md new file mode 100644 index 0000000..cd61279 --- /dev/null +++ b/benchmarks/2026-09-29-follower-receiver-check/README.md @@ -0,0 +1,48 @@ +# Current release check with an opt-in follower store + +This September 29, 2026 (Pacific time) sample checks the release binary built +from PR #14 after adding a persistent follower receiver. It does **not** test +follower-backed commit acknowledgments: the node advertises no follower +capacity, and the serving path still waits for RustFS publication. The binary +SHA-256 was `c75ecbcf8ad2277c5b6ee6356604fb427539edf21d6660ad5c2513a9d16431f3`. + +The Mac14,13 host has 12 logical CPUs and was heavily loaded (one-minute load +observed at 22.8โ€“25.8 around this fixture, rising above 30 during the run). +A fresh RustFS container used the pinned +`ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858` +image. The server used four initial partitions, `auth_cache_enabled: true`, +the default 500 ms routed-page/local-handle caches, and a 1 GiB opt-in +`follower_store_bytes` budget. The benchmark used signed boto3 with zero SDK +retries, 64 seeded 1 KiB items, two items per batch or transaction request, +one/eight closed-loop clients, and five seconds per case. All 24 cases +completed with zero SDK request errors. + +| API | Requests/s, 1 / 8 clients | p95 ms, 1 / 8 clients | +| --- | ---: | ---: | +| GetItem | 272.61 / 768.80 | 9.27 / 22.16 | +| Query | 447.11 / 688.26 | 4.85 / 23.72 | +| Scan | 284.74 / 619.05 | 7.60 / 25.33 | +| BatchGetItem | 210.19 / 480.45 | 10.95 / 33.19 | +| TransactGetItems | 1.31 / 3.35 | 1,250.76 / 3,470.13 | +| DescribeTable | 508.32 / 755.48 | 4.15 / 23.07 | +| ListTables | 623.49 / 620.23 | 3.38 / 28.34 | +| PutItem | 10.91 / 45.55 | 174.28 / 462.72 | +| UpdateItem | 18.97 / 31.03 | 98.15 / 558.94 | +| DeleteItem, absent key | 14.36 / 54.32 | 123.78 / 391.83 | +| BatchWriteItem | 6.04 / 21.95 | 265.89 / 687.50 | +| TransactWriteItems | 1.39 / 4.31 | 837.09 / 2,286.77 | + +BatchGetItem moved 420.39/960.90 items/s and BatchWriteItem moved +12.07/43.91 items/s at one/eight clients. The server logged six deferred +background operations, including Cell mailbox-byte exhaustion and incomplete +transaction recovery. Zero foreground errors therefore do not establish +sustainable throughput. The host load and code revision differ from the old +1,488.6/1,903.2 GetItem sample, so this is another failed reproduction of +those peaks, not a controlled regression measurement. + +The [raw results](run.json) contain every case, elapsed duration, count, +error count, and latency percentile. A signed AWS CLI smoke separately created +a table, wrote and read an item with the follower store configured, then +checked that item after a hard server restart. The follower-lane unit test +covers an authorized append and reopen of the store. Neither test proves that +the serving write path uses follower durability. diff --git a/benchmarks/2026-09-29-follower-receiver-check/run.json b/benchmarks/2026-09-29-follower-receiver-check/run.json new file mode 100644 index 0000000..5e6387c --- /dev/null +++ b/benchmarks/2026-09-29-follower-receiver-check/run.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T02:15:35.883291+00:00", + "endpoint": "http://127.0.0.1:58722", + "table": "FollowerSmoke", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1364, + "errors": 0, + "first_error": null, + "ops_per_s": 272.61, + "items_per_s": 272.61, + "mean_ms": 3.67, + "p50_ms": 2.63, + "p95_ms": 9.27, + "p99_ms": 16.39 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3849, + "errors": 0, + "first_error": null, + "ops_per_s": 768.8, + "items_per_s": 768.8, + "mean_ms": 10.39, + "p50_ms": 8.9, + "p95_ms": 22.16, + "p99_ms": 33.34 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2237, + "errors": 0, + "first_error": null, + "ops_per_s": 447.11, + "items_per_s": 447.11, + "mean_ms": 2.24, + "p50_ms": 1.86, + "p95_ms": 4.85, + "p99_ms": 8.45 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 3444, + "errors": 0, + "first_error": null, + "ops_per_s": 688.26, + "items_per_s": 688.26, + "mean_ms": 11.61, + "p50_ms": 10.14, + "p95_ms": 23.72, + "p99_ms": 32.61 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1424, + "errors": 0, + "first_error": null, + "ops_per_s": 284.74, + "items_per_s": 284.74, + "mean_ms": 3.51, + "p50_ms": 2.7, + "p95_ms": 7.6, + "p99_ms": 12.37 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 3098, + "errors": 0, + "first_error": null, + "ops_per_s": 619.05, + "items_per_s": 619.05, + "mean_ms": 12.89, + "p50_ms": 11.28, + "p95_ms": 25.33, + "p99_ms": 49.08 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.01, + "completed": 1054, + "errors": 0, + "first_error": null, + "ops_per_s": 210.19, + "items_per_s": 420.39, + "mean_ms": 4.75, + "p50_ms": 3.7, + "p95_ms": 10.95, + "p99_ms": 18.24 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.0, + "completed": 2404, + "errors": 0, + "first_error": null, + "ops_per_s": 480.45, + "items_per_s": 960.9, + "mean_ms": 16.63, + "p50_ms": 14.24, + "p95_ms": 33.19, + "p99_ms": 56.28 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.34, + "completed": 7, + "errors": 0, + "first_error": null, + "ops_per_s": 1.31, + "items_per_s": 2.62, + "mean_ms": 762.3, + "p50_ms": 702.61, + "p95_ms": 1250.76, + "p99_ms": 1250.76 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 6.86, + "completed": 23, + "errors": 0, + "first_error": null, + "ops_per_s": 3.35, + "items_per_s": 6.71, + "mean_ms": 2261.5, + "p50_ms": 2481.32, + "p95_ms": 3470.13, + "p99_ms": 3792.83 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2543, + "errors": 0, + "first_error": null, + "ops_per_s": 508.32, + "items_per_s": 0.0, + "mean_ms": 1.97, + "p50_ms": 1.47, + "p95_ms": 4.15, + "p99_ms": 6.98 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3786, + "errors": 0, + "first_error": null, + "ops_per_s": 755.48, + "items_per_s": 0.0, + "mean_ms": 10.57, + "p50_ms": 9.15, + "p95_ms": 23.07, + "p99_ms": 31.47 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 3118, + "errors": 0, + "first_error": null, + "ops_per_s": 623.49, + "items_per_s": 0.0, + "mean_ms": 1.6, + "p50_ms": 1.39, + "p95_ms": 3.38, + "p99_ms": 5.06 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3107, + "errors": 0, + "first_error": null, + "ops_per_s": 620.23, + "items_per_s": 0.0, + "mean_ms": 12.87, + "p50_ms": 10.54, + "p95_ms": 28.34, + "p99_ms": 40.96 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.04, + "completed": 55, + "errors": 0, + "first_error": null, + "ops_per_s": 10.91, + "items_per_s": 10.91, + "mean_ms": 91.64, + "p50_ms": 74.79, + "p95_ms": 174.28, + "p99_ms": 188.99 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.12, + "completed": 233, + "errors": 0, + "first_error": null, + "ops_per_s": 45.55, + "items_per_s": 45.55, + "mean_ms": 174.08, + "p50_ms": 139.32, + "p95_ms": 462.72, + "p99_ms": 611.52 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 95, + "errors": 0, + "first_error": null, + "ops_per_s": 18.97, + "items_per_s": 18.97, + "mean_ms": 52.7, + "p50_ms": 44.1, + "p95_ms": 98.15, + "p99_ms": 248.78 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.22, + "completed": 162, + "errors": 0, + "first_error": null, + "ops_per_s": 31.03, + "items_per_s": 31.03, + "mean_ms": 252.4, + "p50_ms": 199.43, + "p95_ms": 558.94, + "p99_ms": 681.34 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.02, + "completed": 72, + "errors": 0, + "first_error": null, + "ops_per_s": 14.36, + "items_per_s": 14.36, + "mean_ms": 69.65, + "p50_ms": 64.34, + "p95_ms": 123.78, + "p99_ms": 143.88 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.08, + "completed": 276, + "errors": 0, + "first_error": null, + "ops_per_s": 54.32, + "items_per_s": 54.32, + "mean_ms": 146.64, + "p50_ms": 119.13, + "p95_ms": 391.83, + "p99_ms": 489.2 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.14, + "completed": 31, + "errors": 0, + "first_error": null, + "ops_per_s": 6.04, + "items_per_s": 12.07, + "mean_ms": 165.65, + "p50_ms": 150.71, + "p95_ms": 265.89, + "p99_ms": 297.04 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.51, + "completed": 121, + "errors": 0, + "first_error": null, + "ops_per_s": 21.95, + "items_per_s": 43.91, + "mean_ms": 344.1, + "p50_ms": 260.3, + "p95_ms": 687.5, + "p99_ms": 752.02 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.05, + "completed": 7, + "errors": 0, + "first_error": null, + "ops_per_s": 1.39, + "items_per_s": 2.77, + "mean_ms": 720.67, + "p50_ms": 695.67, + "p95_ms": 837.09, + "p99_ms": 837.09 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 6.95, + "completed": 30, + "errors": 0, + "first_error": null, + "ops_per_s": 4.31, + "items_per_s": 8.63, + "mean_ms": 1652.82, + "p50_ms": 1555.06, + "p95_ms": 2286.77, + "p99_ms": 2297.14 + } + ] +} diff --git a/benchmarks/2026-09-29-index-batch-rerun/README.md b/benchmarks/2026-09-29-index-batch-rerun/README.md new file mode 100644 index 0000000..d9f5114 --- /dev/null +++ b/benchmarks/2026-09-29-index-batch-rerun/README.md @@ -0,0 +1,32 @@ +# Index batching release benchmark rerun + +On September 29, 2026 (Pacific time), a release build of BeyondDB at commit `1c38cca32251867236f9c11e09d5ae6fa3526cd4` plus uncommitted recovery wiring and bounded global-index projection ran against pinned ExtendDB SQLite commit `7eaa89b437feed0af0f05883d3f1493f86c6fc6d`. The BeyondDB source diff SHA-256 was `0d0d600c7d90a57485ee24001590b1068042f66fd0ebd96c15f2dd4110cf1244`; the release binary SHA-256 was `1ff8d4525c055bf89b2ffd5e05315c373ca3346b03e41147f48f63b0f522c73f`. Both builds ran sequentially on the same 12-logical-CPU Mac. + +The signed boto3 1.43.105 harness used fresh tables, 64 seeded 1 KiB items, five seconds per case, one and eight closed-loop clients, consistent reads, zero SDK retries, and two items per batch or transaction. `DeleteItem` targeted unique absent keys. Both backends completed all 24 cases with zero foreground SDK errors. + +| API | BeyondDB req/s (1 / 8 clients) | SQLite req/s (1 / 8 clients) | BeyondDB p95 ms (1 / 8 clients) | SQLite p95 ms (1 / 8 clients) | +| --- | ---: | ---: | ---: | ---: | +| GetItem | 170.02 / 616.17 | 349.73 / 518.20 | 16.66 / 29.26 | 9.66 / 35.57 | +| Query | 255.35 / 477.00 | 211.37 / 620.45 | 11.77 / 40.15 | 14.44 / 28.49 | +| Scan | 215.15 / 582.24 | 204.54 / 612.03 | 11.90 / 26.59 | 12.68 / 28.27 | +| BatchGetItem | 117.17 / 520.94 | 106.92 / 510.79 | 22.57 / 33.78 | 26.44 / 35.03 | +| TransactGetItems | 1.84 / 3.79 | 65.12 / 437.19 | 1,012.49 / 2,999.63 | 59.85 / 43.48 | +| DescribeTable | 540.04 / 754.54 | 68.53 / 400.34 | 3.77 / 23.36 | 34.68 / 44.01 | +| ListTables | 348.04 / 607.27 | 237.51 / 586.67 | 5.87 / 29.59 | 13.39 / 29.68 | +| PutItem | 13.35 / 74.99 | 113.02 / 456.42 | 154.15 / 240.39 | 19.08 / 37.85 | +| UpdateItem | 14.49 / 21.97 | 290.76 / 410.65 | 134.92 / 826.87 | 7.93 / 52.18 | +| DeleteItem, absent key | 14.08 / 71.90 | 322.06 / 781.15 | 119.25 / 266.14 | 7.64 / 21.47 | +| BatchWriteItem | 7.39 / 21.47 | 192.13 / 163.82 | 227.62 / 882.15 | 11.48 / 101.78 | +| TransactWriteItems | 1.58 / 3.22 | 122.88 / 262.70 | 643.43 / 2,843.73 | 25.32 / 86.60 | + +BatchGetItem completed 234.34 / 1,041.88 items/s in BeyondDB and 213.84 / 1,021.58 items/s in SQLite. BatchWriteItem completed 14.78 / 42.94 items/s in BeyondDB and 384.26 / 327.64 items/s in SQLite. + +## Fixture and limits + +- BeyondDB used four initial partitions, twelve SQL workers, 128 active Cell slots, `auth_cache_enabled: true`, a 1 GiB follower-store budget, and a fresh RustFS container at the pinned `ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858` image. Follower-backed commit proof remains disabled, so writes await object-store publication. IAM was verified. +- ExtendDB used file-backed SQLite in `sqlite,dev-mode`. SigV4 was verified; authorization was open in dev mode. Its local-file durability and authorization differ from BeyondDB. +- One-minute host load was 27.04 to 21.58 during BeyondDB and 21.91 to 26.99 during SQLite, against 12 logical CPUs. These contended, sequential five-second cases cannot establish a stable speed ratio or fleet-scale capacity. The global-index batching change targets background projection; this workload did not create or query a GSI. +- BeyondDB logged one deferred transaction recovery pass: `cross-Cell participant resolution is incomplete`. Zero foreground SDK errors do not prove all background work has caught up. The transaction and write gaps remain substantial, and this run does not meet the performance objective. +- The source tree was uncommitted when the binary was built. This report describes that exact binary, not a clean release commit. + +See [BeyondDB raw cases](run.json), [SQLite raw cases](sqlite-run.json), [BeyondDB metadata](meta.json), [SQLite metadata](sqlite-meta.json), and [BeyondDB warnings](server-warnings.txt). diff --git a/benchmarks/2026-09-29-index-batch-rerun/meta.json b/benchmarks/2026-09-29-index-batch-rerun/meta.json new file mode 100644 index 0000000..e7a0f45 --- /dev/null +++ b/benchmarks/2026-09-29-index-batch-rerun/meta.json @@ -0,0 +1,29 @@ +{ + "server_commit": "1c38cca + uncommitted recovery wiring and index batching", + "binary_sha256": "1ff8d4525c055bf89b2ffd5e05315c373ca3346b03e41147f48f63b0f522c73f", + "rustfs_image": "ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858", + "host_load_start": [ + 27.0361328125, + 26.0625, + 22.32763671875 + ], + "logical_cpus": 12, + "boto3_version": "1.43.105", + "fixture": { + "initial_partitions": 4, + "sql_workers": 12, + "max_active_cells": 128, + "follower_store_bytes": 1073741824, + "auth_cache_enabled": true, + "server_object_publication": true + }, + "endpoint": "http://127.0.0.1:54679", + "started_at_unix": 1790743516.3009431, + "host_load_end": [ + 21.58349609375, + 24.58544921875, + 22.26171875 + ], + "ended_at_unix": 1790743646.1916249, + "bench_exit_code": 0 +} diff --git a/benchmarks/2026-09-29-index-batch-rerun/run.json b/benchmarks/2026-09-29-index-batch-rerun/run.json new file mode 100644 index 0000000..36905aa --- /dev/null +++ b/benchmarks/2026-09-29-index-batch-rerun/run.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T04:45:21.144525+00:00", + "endpoint": "http://127.0.0.1:54679", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 851, + "errors": 0, + "first_error": null, + "ops_per_s": 170.02, + "items_per_s": 170.02, + "mean_ms": 5.88, + "p50_ms": 4.14, + "p95_ms": 16.66, + "p99_ms": 28.22 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3089, + "errors": 0, + "first_error": null, + "ops_per_s": 616.17, + "items_per_s": 616.17, + "mean_ms": 12.97, + "p50_ms": 10.84, + "p95_ms": 29.26, + "p99_ms": 43.2 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 1279, + "errors": 0, + "first_error": null, + "ops_per_s": 255.35, + "items_per_s": 255.35, + "mean_ms": 3.91, + "p50_ms": 2.45, + "p95_ms": 11.77, + "p99_ms": 21.86 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2390, + "errors": 0, + "first_error": null, + "ops_per_s": 477.0, + "items_per_s": 477.0, + "mean_ms": 16.73, + "p50_ms": 13.18, + "p95_ms": 40.15, + "p99_ms": 69.82 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 1077, + "errors": 0, + "first_error": null, + "ops_per_s": 215.15, + "items_per_s": 215.15, + "mean_ms": 4.64, + "p50_ms": 2.96, + "p95_ms": 11.9, + "p99_ms": 29.56 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2919, + "errors": 0, + "first_error": null, + "ops_per_s": 582.24, + "items_per_s": 582.24, + "mean_ms": 13.7, + "p50_ms": 11.46, + "p95_ms": 26.59, + "p99_ms": 52.34 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 586, + "errors": 0, + "first_error": null, + "ops_per_s": 117.17, + "items_per_s": 234.33, + "mean_ms": 8.53, + "p50_ms": 5.98, + "p95_ms": 22.57, + "p99_ms": 74.03 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.03, + "completed": 2622, + "errors": 0, + "first_error": null, + "ops_per_s": 520.94, + "items_per_s": 1041.89, + "mean_ms": 15.29, + "p50_ms": 12.92, + "p95_ms": 33.78, + "p99_ms": 48.76 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.43, + "completed": 10, + "errors": 0, + "first_error": null, + "ops_per_s": 1.84, + "items_per_s": 3.68, + "mean_ms": 543.06, + "p50_ms": 542.41, + "p95_ms": 1012.49, + "p99_ms": 1012.49 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.81, + "completed": 22, + "errors": 0, + "first_error": null, + "ops_per_s": 3.79, + "items_per_s": 7.57, + "mean_ms": 1925.51, + "p50_ms": 2297.74, + "p95_ms": 2999.63, + "p99_ms": 3116.81 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2701, + "errors": 0, + "first_error": null, + "ops_per_s": 540.04, + "items_per_s": 0.0, + "mean_ms": 1.85, + "p50_ms": 1.57, + "p95_ms": 3.77, + "p99_ms": 6.5 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3777, + "errors": 0, + "first_error": null, + "ops_per_s": 754.54, + "items_per_s": 0.0, + "mean_ms": 10.59, + "p50_ms": 8.47, + "p95_ms": 23.36, + "p99_ms": 41.17 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1741, + "errors": 0, + "first_error": null, + "ops_per_s": 348.04, + "items_per_s": 0.0, + "mean_ms": 2.87, + "p50_ms": 2.19, + "p95_ms": 5.87, + "p99_ms": 13.56 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3040, + "errors": 0, + "first_error": null, + "ops_per_s": 607.27, + "items_per_s": 0.0, + "mean_ms": 13.16, + "p50_ms": 10.81, + "p95_ms": 29.59, + "p99_ms": 45.9 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.02, + "completed": 67, + "errors": 0, + "first_error": null, + "ops_per_s": 13.35, + "items_per_s": 13.35, + "mean_ms": 74.91, + "p50_ms": 68.64, + "p95_ms": 154.15, + "p99_ms": 170.93 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.13, + "completed": 385, + "errors": 0, + "first_error": null, + "ops_per_s": 74.99, + "items_per_s": 74.99, + "mean_ms": 105.91, + "p50_ms": 87.96, + "p95_ms": 240.39, + "p99_ms": 298.12 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.04, + "completed": 73, + "errors": 0, + "first_error": null, + "ops_per_s": 14.49, + "items_per_s": 14.49, + "mean_ms": 69.01, + "p50_ms": 60.7, + "p95_ms": 134.92, + "p99_ms": 169.98 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.32, + "completed": 117, + "errors": 0, + "first_error": null, + "ops_per_s": 21.97, + "items_per_s": 21.97, + "mean_ms": 351.86, + "p50_ms": 285.33, + "p95_ms": 826.87, + "p99_ms": 1141.16 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.04, + "completed": 71, + "errors": 0, + "first_error": null, + "ops_per_s": 14.08, + "items_per_s": 14.08, + "mean_ms": 71.01, + "p50_ms": 63.79, + "p95_ms": 119.25, + "p99_ms": 201.75 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.22, + "completed": 375, + "errors": 0, + "first_error": null, + "ops_per_s": 71.9, + "items_per_s": 71.9, + "mean_ms": 108.01, + "p50_ms": 83.5, + "p95_ms": 266.14, + "p99_ms": 346.35 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.14, + "completed": 38, + "errors": 0, + "first_error": null, + "ops_per_s": 7.39, + "items_per_s": 14.77, + "mean_ms": 135.35, + "p50_ms": 108.95, + "p95_ms": 227.62, + "p99_ms": 233.53 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.31, + "completed": 114, + "errors": 0, + "first_error": null, + "ops_per_s": 21.47, + "items_per_s": 42.95, + "mean_ms": 363.85, + "p50_ms": 306.39, + "p95_ms": 882.15, + "p99_ms": 1038.94 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.05, + "completed": 8, + "errors": 0, + "first_error": null, + "ops_per_s": 1.58, + "items_per_s": 3.17, + "mean_ms": 631.56, + "p50_ms": 528.73, + "p95_ms": 643.43, + "p99_ms": 643.43 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 7.15, + "completed": 23, + "errors": 0, + "first_error": null, + "ops_per_s": 3.22, + "items_per_s": 6.44, + "mean_ms": 2166.26, + "p50_ms": 2214.84, + "p95_ms": 2843.73, + "p99_ms": 2947.2 + } + ] +} diff --git a/benchmarks/2026-09-29-index-batch-rerun/server-warnings.txt b/benchmarks/2026-09-29-index-batch-rerun/server-warnings.txt new file mode 100644 index 0000000..7b1a2ec --- /dev/null +++ b/benchmarks/2026-09-29-index-batch-rerun/server-warnings.txt @@ -0,0 +1 @@ +2026-09-30T04:46:11.756232Z WARN beyonddb::provision::transactions: transaction recovery deferred error=cross-Cell participant resolution is incomplete diff --git a/benchmarks/2026-09-29-index-batch-rerun/sqlite-meta.json b/benchmarks/2026-09-29-index-batch-rerun/sqlite-meta.json new file mode 100644 index 0000000..cd6ee8e --- /dev/null +++ b/benchmarks/2026-09-29-index-batch-rerun/sqlite-meta.json @@ -0,0 +1,24 @@ +{ + "server_commit": "7eaa89b437feed0af0f05883d3f1493f86c6fc6d", + "binary_sha256": "dfc9cd868d2bc06b71fa9fbc460a074dca330cb8ceffaebf62d8d54bdb485e38", + "host_load_start": [ + 21.90673828125, + 24.55078125, + 22.27587890625 + ], + "logical_cpus": 12, + "boto3_version": "1.43.105", + "fixture": { + "sqlite_file": "/var/folders/rn/k2b7mvbs2dgb__6ys6_h94pm0000gn/T/extenddb-perf-rerun-pnui8dq0/data.sqlite", + "dev_mode_open_auth": true + }, + "endpoint": "http://127.0.0.1:55036", + "started_at_unix": 1790743653.145322, + "host_load_end": [ + 26.9921875, + 26.15087890625, + 23.2412109375 + ], + "ended_at_unix": 1790743774.472655, + "bench_exit_code": 0 +} diff --git a/benchmarks/2026-09-29-index-batch-rerun/sqlite-run.json b/benchmarks/2026-09-29-index-batch-rerun/sqlite-run.json new file mode 100644 index 0000000..2eafb7c --- /dev/null +++ b/benchmarks/2026-09-29-index-batch-rerun/sqlite-run.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T04:47:33.743596+00:00", + "endpoint": "http://127.0.0.1:55036", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1749, + "errors": 0, + "first_error": null, + "ops_per_s": 349.73, + "items_per_s": 349.73, + "mean_ms": 2.86, + "p50_ms": 1.7, + "p95_ms": 9.66, + "p99_ms": 17.47 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2596, + "errors": 0, + "first_error": null, + "ops_per_s": 518.2, + "items_per_s": 518.2, + "mean_ms": 15.41, + "p50_ms": 12.14, + "p95_ms": 35.57, + "p99_ms": 63.45 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1057, + "errors": 0, + "first_error": null, + "ops_per_s": 211.37, + "items_per_s": 211.37, + "mean_ms": 4.73, + "p50_ms": 2.88, + "p95_ms": 14.44, + "p99_ms": 21.88 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.02, + "completed": 3115, + "errors": 0, + "first_error": null, + "ops_per_s": 620.45, + "items_per_s": 620.45, + "mean_ms": 12.87, + "p50_ms": 10.89, + "p95_ms": 28.49, + "p99_ms": 40.8 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1023, + "errors": 0, + "first_error": null, + "ops_per_s": 204.54, + "items_per_s": 204.54, + "mean_ms": 4.89, + "p50_ms": 3.7, + "p95_ms": 12.68, + "p99_ms": 19.99 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3066, + "errors": 0, + "first_error": null, + "ops_per_s": 612.03, + "items_per_s": 612.03, + "mean_ms": 13.04, + "p50_ms": 10.62, + "p95_ms": 28.27, + "p99_ms": 42.6 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.02, + "completed": 537, + "errors": 0, + "first_error": null, + "ops_per_s": 106.92, + "items_per_s": 213.83, + "mean_ms": 9.35, + "p50_ms": 6.68, + "p95_ms": 26.44, + "p99_ms": 44.58 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.02, + "completed": 2562, + "errors": 0, + "first_error": null, + "ops_per_s": 510.79, + "items_per_s": 1021.58, + "mean_ms": 15.57, + "p50_ms": 12.85, + "p95_ms": 35.03, + "p99_ms": 50.55 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.01, + "completed": 326, + "errors": 0, + "first_error": null, + "ops_per_s": 65.12, + "items_per_s": 130.24, + "mean_ms": 15.34, + "p50_ms": 4.86, + "p95_ms": 59.85, + "p99_ms": 126.46 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.02, + "completed": 2193, + "errors": 0, + "first_error": null, + "ops_per_s": 437.19, + "items_per_s": 874.37, + "mean_ms": 18.22, + "p50_ms": 14.87, + "p95_ms": 43.48, + "p99_ms": 64.46 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.03, + "completed": 345, + "errors": 0, + "first_error": null, + "ops_per_s": 68.53, + "items_per_s": 0.0, + "mean_ms": 14.56, + "p50_ms": 11.57, + "p95_ms": 34.68, + "p99_ms": 49.34 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2005, + "errors": 0, + "first_error": null, + "ops_per_s": 400.34, + "items_per_s": 0.0, + "mean_ms": 19.95, + "p50_ms": 17.02, + "p95_ms": 44.01, + "p99_ms": 62.62 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1188, + "errors": 0, + "first_error": null, + "ops_per_s": 237.51, + "items_per_s": 0.0, + "mean_ms": 4.21, + "p50_ms": 2.51, + "p95_ms": 13.39, + "p99_ms": 26.35 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.03, + "completed": 2948, + "errors": 0, + "first_error": null, + "ops_per_s": 586.67, + "items_per_s": 0.0, + "mean_ms": 13.57, + "p50_ms": 11.64, + "p95_ms": 29.68, + "p99_ms": 43.98 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 566, + "errors": 0, + "first_error": null, + "ops_per_s": 113.02, + "items_per_s": 113.02, + "mean_ms": 8.85, + "p50_ms": 7.27, + "p95_ms": 19.08, + "p99_ms": 33.92 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 2283, + "errors": 0, + "first_error": null, + "ops_per_s": 456.42, + "items_per_s": 456.42, + "mean_ms": 17.52, + "p50_ms": 14.81, + "p95_ms": 37.85, + "p99_ms": 54.68 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1454, + "errors": 0, + "first_error": null, + "ops_per_s": 290.76, + "items_per_s": 290.76, + "mean_ms": 3.44, + "p50_ms": 2.7, + "p95_ms": 7.93, + "p99_ms": 12.83 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.09, + "completed": 2090, + "errors": 0, + "first_error": null, + "ops_per_s": 410.65, + "items_per_s": 410.65, + "mean_ms": 19.23, + "p50_ms": 14.37, + "p95_ms": 52.18, + "p99_ms": 95.21 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1611, + "errors": 0, + "first_error": null, + "ops_per_s": 322.06, + "items_per_s": 322.06, + "mean_ms": 3.1, + "p50_ms": 2.3, + "p95_ms": 7.64, + "p99_ms": 12.27 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3912, + "errors": 0, + "first_error": null, + "ops_per_s": 781.15, + "items_per_s": 781.15, + "mean_ms": 10.22, + "p50_ms": 8.99, + "p95_ms": 21.47, + "p99_ms": 31.15 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 961, + "errors": 0, + "first_error": null, + "ops_per_s": 192.13, + "items_per_s": 384.26, + "mean_ms": 5.2, + "p50_ms": 4.0, + "p95_ms": 11.48, + "p99_ms": 18.99 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.04, + "completed": 825, + "errors": 0, + "first_error": null, + "ops_per_s": 163.82, + "items_per_s": 327.63, + "mean_ms": 48.7, + "p50_ms": 37.25, + "p95_ms": 101.78, + "p99_ms": 183.89 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 615, + "errors": 0, + "first_error": null, + "ops_per_s": 122.88, + "items_per_s": 245.77, + "mean_ms": 8.14, + "p50_ms": 5.01, + "p95_ms": 25.32, + "p99_ms": 43.08 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.01, + "completed": 1317, + "errors": 0, + "first_error": null, + "ops_per_s": 262.7, + "items_per_s": 525.4, + "mean_ms": 30.39, + "p50_ms": 20.1, + "p95_ms": 86.6, + "p99_ms": 126.35 + } + ] +} diff --git a/benchmarks/2026-09-29-metadata-cache/README.md b/benchmarks/2026-09-29-metadata-cache/README.md new file mode 100644 index 0000000..78a3f71 --- /dev/null +++ b/benchmarks/2026-09-29-metadata-cache/README.md @@ -0,0 +1,35 @@ +# Opt-in metadata response cache sample + +A release binary built from `fef8da8` plus the metadata-cache working-tree +change ran against a fresh four-partition RustFS fixture. It used +`auth_cache_enabled: true`, signed boto3, 64 seeded items with 1 KiB payloads, +no SDK retries, and five seconds per case. The `DescribeTable` and +`ListTables` responses were cached for 500 ms on this node. Local +`CreateTable`, `DeleteTable`, and `UpdateTable` commits invalidate the cache; +a signed SDK integration test checked immediate visibility after create and +update. The four benchmark cases completed with zero SDK request errors and +no server warnings. + +A fresh file-backed ExtendDB SQLite `7eaa89b` development-mode fixture ran +the same two operations and client counts. It verified SigV4 but used open +authorization. The earlier BeyondDB column comes from the full +[12-API comparison](../2026-09-29-sqlite-comparison/README.md) before this +cache change; it is a separate fixture, not a controlled A/B run. + +| API | BeyondDB before, requests/s 1 / 8 | BeyondDB with cache, 1 / 8 | SQLite fresh, 1 / 8 | +| --- | ---: | ---: | ---: | +| DescribeTable | 285.22 / 384.20 | 488.70 / 571.05 | 835.34 / 1,196.07 | +| ListTables | 214.85 / 472.34 | 582.96 / 785.79 | 700.59 / 342.71 | + +The one-minute host load average was 27.8 to 32.8 during the cache fixture +and 29.2 to 28.8 during the SQLite fixture on 12 logical CPUs. Other virtual +machines remained active. The cached metadata cases improved over the earlier +BeyondDB sample, but `DescribeTable` and one-client `ListTables` remained below +SQLite. The eight-client SQLite `ListTables` result fell sharply under host +contention and is not a reliable parity target. This is a short local +diagnostic, not sustainable throughput or a fleet sizing claim. + +Raw results: [BeyondDB](beyonddb.json), [ExtendDB SQLite](extenddb-sqlite.json), +and [BeyondDB warnings](beyonddb-warnings.txt) (empty). Both fixtures used +the same `scripts/bench.py` call shape with +`--seconds 5 --clients 1 8 --payload-bytes 1024 --operations describe_table list_tables`. diff --git a/benchmarks/2026-09-29-metadata-cache/beyonddb-warnings.txt b/benchmarks/2026-09-29-metadata-cache/beyonddb-warnings.txt new file mode 100644 index 0000000..e69de29 diff --git a/benchmarks/2026-09-29-metadata-cache/beyonddb.json b/benchmarks/2026-09-29-metadata-cache/beyonddb.json new file mode 100644 index 0000000..00ab28a --- /dev/null +++ b/benchmarks/2026-09-29-metadata-cache/beyonddb.json @@ -0,0 +1,77 @@ +{ + "started_at": "2026-09-30T00:19:32.717111+00:00", + "endpoint": "http://127.0.0.1:59627", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "describe_table", + "list_tables" + ], + "cases": [ + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2444, + "errors": 0, + "first_error": null, + "ops_per_s": 488.7, + "items_per_s": 0.0, + "mean_ms": 2.04, + "p50_ms": 1.79, + "p95_ms": 4.24, + "p99_ms": 7.37 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2859, + "errors": 0, + "first_error": null, + "ops_per_s": 571.05, + "items_per_s": 0.0, + "mean_ms": 14.0, + "p50_ms": 11.34, + "p95_ms": 31.82, + "p99_ms": 48.17 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2915, + "errors": 0, + "first_error": null, + "ops_per_s": 582.96, + "items_per_s": 0.0, + "mean_ms": 1.71, + "p50_ms": 1.53, + "p95_ms": 3.21, + "p99_ms": 5.51 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3933, + "errors": 0, + "first_error": null, + "ops_per_s": 785.79, + "items_per_s": 0.0, + "mean_ms": 10.17, + "p50_ms": 8.65, + "p95_ms": 22.23, + "p99_ms": 33.36 + } + ] +} diff --git a/benchmarks/2026-09-29-metadata-cache/extenddb-sqlite.json b/benchmarks/2026-09-29-metadata-cache/extenddb-sqlite.json new file mode 100644 index 0000000..8b6b47e --- /dev/null +++ b/benchmarks/2026-09-29-metadata-cache/extenddb-sqlite.json @@ -0,0 +1,77 @@ +{ + "started_at": "2026-09-30T00:20:06.151000+00:00", + "endpoint": "http://127.0.0.1:59705", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "describe_table", + "list_tables" + ], + "cases": [ + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 4177, + "errors": 0, + "first_error": null, + "ops_per_s": 835.34, + "items_per_s": 0.0, + "mean_ms": 1.2, + "p50_ms": 1.03, + "p95_ms": 2.22, + "p99_ms": 3.74 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.0, + "completed": 5985, + "errors": 0, + "first_error": null, + "ops_per_s": 1196.07, + "items_per_s": 0.0, + "mean_ms": 6.68, + "p50_ms": 6.14, + "p95_ms": 12.16, + "p99_ms": 17.22 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.03, + "completed": 3521, + "errors": 0, + "first_error": null, + "ops_per_s": 700.59, + "items_per_s": 0.0, + "mean_ms": 1.42, + "p50_ms": 1.26, + "p95_ms": 2.59, + "p99_ms": 4.3 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.05, + "completed": 1732, + "errors": 0, + "first_error": null, + "ops_per_s": 342.71, + "items_per_s": 0.0, + "mean_ms": 23.2, + "p50_ms": 14.71, + "p95_ms": 79.26, + "p99_ms": 134.33 + } + ] +} diff --git a/benchmarks/2026-09-29-recovery-wiring-rerun/README.md b/benchmarks/2026-09-29-recovery-wiring-rerun/README.md new file mode 100644 index 0000000..2225301 --- /dev/null +++ b/benchmarks/2026-09-29-recovery-wiring-rerun/README.md @@ -0,0 +1,32 @@ +# Recovery wiring release benchmark rerun + +On September 29, 2026 (Pacific time), a release build of BeyondDB at commit `1c38cca32251867236f9c11e09d5ae6fa3526cd4` plus the uncommitted recovery wiring (working-tree diff SHA-256 `276cf032a9bf92c3340d0d435ae5e0c91552170163f94affbe4c34b761fab469`) ran the same signed boto3 workload as pinned ExtendDB SQLite commit `7eaa89b437feed0af0f05883d3f1493f86c6fc6d`. The BeyondDB binary SHA-256 was `6210b02f6a0bd3d9c776da5c227a5818a8a55d798c873dc8e6fc777261cf5c89`. Both backends ran sequentially on the same 12-logical-CPU Mac. + +The harness used fresh tables, 64 seeded 1 KiB items, five seconds per case, one and eight closed-loop clients, consistent reads, signed boto3 1.43.105 requests with zero retries, and two items per batch or transaction. `DeleteItem` targeted unique absent keys. Both backends completed all 24 cases with zero foreground SDK errors. + +| API | BeyondDB req/s (1 / 8 clients) | SQLite req/s (1 / 8 clients) | BeyondDB p95 ms (1 / 8 clients) | SQLite p95 ms (1 / 8 clients) | +| --- | ---: | ---: | ---: | ---: | +| GetItem | 453.35 / 650.82 | 423.14 / 772.65 | 4.41 / 26.01 | 6.20 / 21.55 | +| Query | 225.86 / 522.46 | 348.79 / 615.96 | 11.12 / 33.69 | 6.85 / 26.94 | +| Scan | 211.00 / 673.24 | 386.09 / 649.88 | 12.93 / 24.91 | 6.05 / 25.70 | +| BatchGetItem | 164.66 / 519.08 | 253.78 / 641.22 | 14.31 / 34.41 | 10.14 / 26.85 | +| TransactGetItems | 1.26 / 4.32 | 182.98 / 541.11 | 1,081.00 / 2,627.54 | 14.52 / 32.02 | +| DescribeTable | 541.96 / 681.30 | 381.88 / 557.12 | 3.38 / 25.84 | 6.22 / 28.81 | +| ListTables | 518.50 / 753.05 | 305.57 / 860.62 | 3.80 / 23.04 | 7.48 / 19.48 | +| PutItem | 12.19 / 56.05 | 305.12 / 317.73 | 149.16 / 431.66 | 8.34 / 58.22 | +| UpdateItem | 17.56 / 28.87 | 184.41 / 349.01 | 122.85 / 588.63 | 13.50 / 50.13 | +| DeleteItem, absent key | 14.42 / 66.58 | 209.71 / 344.37 | 101.96 / 301.83 | 11.91 / 55.85 | +| BatchWriteItem | 8.72 / 16.25 | 96.72 / 159.27 | 224.60 / 1,004.40 | 24.59 / 107.01 | +| TransactWriteItems | 0.92 / 2.66 | 67.94 / 215.97 | 1,148.92 / 3,883.23 | 33.46 / 78.71 | + +BatchGetItem completed 329.33 / 1,038.17 items/s in BeyondDB and 507.56 / 1,282.44 items/s in SQLite. BatchWriteItem completed 17.44 / 32.50 items/s in BeyondDB and 193.44 / 318.54 items/s in SQLite. + +## Fixture and interpretation + +- BeyondDB used four initial partitions, twelve SQL workers, 128 active Cell slots, `auth_cache_enabled: true`, a 1 GiB persistent follower-store budget, and fresh RustFS at the pinned `ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858` image. Follower-backed commit proof remains disabled; writes await object-store publication. IAM was verified. +- ExtendDB used file-backed SQLite `sqlite,dev-mode`. SigV4 was verified, while authorization was open in dev mode. Local-file durability and authorization therefore differ from BeyondDB. +- Host load averaged 26.03 to 28.55 during BeyondDB and 28.40 to 26.95 during SQLite on 12 logical CPUs. These heavily contended, sequential runs do not establish a stable causal speed ratio or production capacity. +- BeyondDB logged four deferred background operations due to Cell mailbox-byte exhaustion: two global-index projections, one capacity sweep, and one routed stream retention sweep. Zero foreground SDK errors do not prove sustainable throughput while background work falls behind. +- The BeyondDB binary includes uncommitted recovery wiring that does not enable follower-backed commits. This run is a functional and performance check of that exact working tree; it is not a benchmark of a clean release commit. + +See [BeyondDB raw cases](run.json), [SQLite raw cases](sqlite-run.json), [BeyondDB metadata](meta.json), [SQLite metadata](sqlite-meta.json), and [BeyondDB warnings](server-warnings.txt). diff --git a/benchmarks/2026-09-29-recovery-wiring-rerun/meta.json b/benchmarks/2026-09-29-recovery-wiring-rerun/meta.json new file mode 100644 index 0000000..c8f1303 --- /dev/null +++ b/benchmarks/2026-09-29-recovery-wiring-rerun/meta.json @@ -0,0 +1,30 @@ +{ + "server_commit": "1c38cca + uncommitted recovery wiring", + "binary_sha256": "6210b02f6a0bd3d9c776da5c227a5818a8a55d798c873dc8e6fc777261cf5c89", + "rustfs_image": "ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858", + "host_load_start": [ + 26.0283203125, + 27.23828125, + 25.69677734375 + ], + "logical_cpus": 12, + "boto3_version": "1.43.105", + "fixture": { + "initial_partitions": 4, + "sql_workers": 12, + "max_active_cells": 128, + "follower_store_bytes": 1073741824, + "auth_cache_enabled": true, + "server_object_publication": true + }, + "endpoint": "http://127.0.0.1:64941", + "started_at_unix": 1790740269.114651, + "host_load_end": [ + 28.5517578125, + 28.056640625, + 26.251953125 + ], + "ended_at_unix": 1790740398.139356, + "bench_exit_code": 0, + "working_tree_diff_sha256": "276cf032a9bf92c3340d0d435ae5e0c91552170163f94affbe4c34b761fab469" +} diff --git a/benchmarks/2026-09-29-recovery-wiring-rerun/run.json b/benchmarks/2026-09-29-recovery-wiring-rerun/run.json new file mode 100644 index 0000000..1a18100 --- /dev/null +++ b/benchmarks/2026-09-29-recovery-wiring-rerun/run.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T03:51:12.478508+00:00", + "endpoint": "http://127.0.0.1:64941", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2267, + "errors": 0, + "first_error": null, + "ops_per_s": 453.35, + "items_per_s": 453.35, + "mean_ms": 2.2, + "p50_ms": 1.93, + "p95_ms": 4.41, + "p99_ms": 8.23 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3260, + "errors": 0, + "first_error": null, + "ops_per_s": 650.82, + "items_per_s": 650.82, + "mean_ms": 12.26, + "p50_ms": 10.69, + "p95_ms": 26.01, + "p99_ms": 34.83 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1130, + "errors": 0, + "first_error": null, + "ops_per_s": 225.86, + "items_per_s": 225.86, + "mean_ms": 4.42, + "p50_ms": 3.27, + "p95_ms": 11.12, + "p99_ms": 19.19 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 2614, + "errors": 0, + "first_error": null, + "ops_per_s": 522.46, + "items_per_s": 522.46, + "mean_ms": 15.29, + "p50_ms": 12.96, + "p95_ms": 33.69, + "p99_ms": 48.69 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1056, + "errors": 0, + "first_error": null, + "ops_per_s": 211.0, + "items_per_s": 211.0, + "mean_ms": 4.74, + "p50_ms": 3.29, + "p95_ms": 12.93, + "p99_ms": 27.79 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3376, + "errors": 0, + "first_error": null, + "ops_per_s": 673.24, + "items_per_s": 673.24, + "mean_ms": 11.85, + "p50_ms": 9.92, + "p95_ms": 24.91, + "p99_ms": 38.17 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 824, + "errors": 0, + "first_error": null, + "ops_per_s": 164.66, + "items_per_s": 329.33, + "mean_ms": 6.07, + "p50_ms": 4.28, + "p95_ms": 14.31, + "p99_ms": 29.56 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2601, + "errors": 0, + "first_error": null, + "ops_per_s": 519.08, + "items_per_s": 1038.17, + "mean_ms": 15.38, + "p50_ms": 12.88, + "p95_ms": 34.41, + "p99_ms": 53.94 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.54, + "completed": 7, + "errors": 0, + "first_error": null, + "ops_per_s": 1.26, + "items_per_s": 2.53, + "mean_ms": 791.87, + "p50_ms": 871.48, + "p95_ms": 1081.0, + "p99_ms": 1081.0 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 6.02, + "completed": 26, + "errors": 0, + "first_error": null, + "ops_per_s": 4.32, + "items_per_s": 8.64, + "mean_ms": 1657.4, + "p50_ms": 1932.47, + "p95_ms": 2627.54, + "p99_ms": 2640.23 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2710, + "errors": 0, + "first_error": null, + "ops_per_s": 541.96, + "items_per_s": 0.0, + "mean_ms": 1.84, + "p50_ms": 1.65, + "p95_ms": 3.38, + "p99_ms": 5.69 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3412, + "errors": 0, + "first_error": null, + "ops_per_s": 681.3, + "items_per_s": 0.0, + "mean_ms": 11.72, + "p50_ms": 10.15, + "p95_ms": 25.84, + "p99_ms": 36.85 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2593, + "errors": 0, + "first_error": null, + "ops_per_s": 518.5, + "items_per_s": 0.0, + "mean_ms": 1.93, + "p50_ms": 1.7, + "p95_ms": 3.8, + "p99_ms": 6.71 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3771, + "errors": 0, + "first_error": null, + "ops_per_s": 753.05, + "items_per_s": 0.0, + "mean_ms": 10.61, + "p50_ms": 9.17, + "p95_ms": 23.04, + "p99_ms": 32.78 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.09, + "completed": 62, + "errors": 0, + "first_error": null, + "ops_per_s": 12.19, + "items_per_s": 12.19, + "mean_ms": 82.04, + "p50_ms": 71.92, + "p95_ms": 149.16, + "p99_ms": 219.23 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.07, + "completed": 284, + "errors": 0, + "first_error": null, + "ops_per_s": 56.05, + "items_per_s": 56.05, + "mean_ms": 142.04, + "p50_ms": 101.8, + "p95_ms": 431.66, + "p99_ms": 536.46 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.07, + "completed": 89, + "errors": 0, + "first_error": null, + "ops_per_s": 17.56, + "items_per_s": 17.56, + "mean_ms": 56.95, + "p50_ms": 43.8, + "p95_ms": 122.85, + "p99_ms": 173.43 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.4, + "completed": 156, + "errors": 0, + "first_error": null, + "ops_per_s": 28.87, + "items_per_s": 28.87, + "mean_ms": 266.46, + "p50_ms": 260.36, + "p95_ms": 588.63, + "p99_ms": 656.45 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.06, + "completed": 73, + "errors": 0, + "first_error": null, + "ops_per_s": 14.42, + "items_per_s": 14.42, + "mean_ms": 69.35, + "p50_ms": 52.11, + "p95_ms": 101.96, + "p99_ms": 302.74 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.05, + "completed": 336, + "errors": 0, + "first_error": null, + "ops_per_s": 66.58, + "items_per_s": 66.58, + "mean_ms": 119.56, + "p50_ms": 88.67, + "p95_ms": 301.83, + "p99_ms": 469.77 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.05, + "completed": 44, + "errors": 0, + "first_error": null, + "ops_per_s": 8.72, + "items_per_s": 17.44, + "mean_ms": 114.68, + "p50_ms": 83.01, + "p95_ms": 224.6, + "p99_ms": 245.78 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.48, + "completed": 89, + "errors": 0, + "first_error": null, + "ops_per_s": 16.25, + "items_per_s": 32.5, + "mean_ms": 478.77, + "p50_ms": 384.4, + "p95_ms": 1004.4, + "p99_ms": 1335.79 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.45, + "completed": 5, + "errors": 0, + "first_error": null, + "ops_per_s": 0.92, + "items_per_s": 1.83, + "mean_ms": 1090.11, + "p50_ms": 1027.12, + "p95_ms": 1148.92, + "p99_ms": 1148.92 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 7.14, + "completed": 19, + "errors": 0, + "first_error": null, + "ops_per_s": 2.66, + "items_per_s": 5.32, + "mean_ms": 2614.44, + "p50_ms": 2413.18, + "p95_ms": 3883.23, + "p99_ms": 3883.23 + } + ] +} diff --git a/benchmarks/2026-09-29-recovery-wiring-rerun/server-warnings.txt b/benchmarks/2026-09-29-recovery-wiring-rerun/server-warnings.txt new file mode 100644 index 0000000..5c4d268 --- /dev/null +++ b/benchmarks/2026-09-29-recovery-wiring-rerun/server-warnings.txt @@ -0,0 +1,4 @@ +2026-09-30T03:51:47.955809Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T03:51:50.346167Z WARN beyonddb::provision::capacity: capacity sweep deferred account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T03:51:51.347233Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T03:52:44.348114Z WARN beyonddb: routed stream retention sweep failed error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes diff --git a/benchmarks/2026-09-29-recovery-wiring-rerun/sqlite-meta.json b/benchmarks/2026-09-29-recovery-wiring-rerun/sqlite-meta.json new file mode 100644 index 0000000..35ee38b --- /dev/null +++ b/benchmarks/2026-09-29-recovery-wiring-rerun/sqlite-meta.json @@ -0,0 +1,24 @@ +{ + "server_commit": "7eaa89b437feed0af0f05883d3f1493f86c6fc6d", + "binary_sha256": "dfc9cd868d2bc06b71fa9fbc460a074dca330cb8ceffaebf62d8d54bdb485e38", + "host_load_start": [ + 28.3994140625, + 28.03857421875, + 26.26611328125 + ], + "logical_cpus": 12, + "boto3_version": "1.43.105", + "fixture": { + "sqlite_file": "/var/folders/rn/k2b7mvbs2dgb__6ys6_h94pm0000gn/T/extenddb-perf-rerun-y2xxqzvu/data.sqlite", + "dev_mode_open_auth": true + }, + "endpoint": "http://127.0.0.1:65298", + "started_at_unix": 1790740408.554809, + "host_load_end": [ + 26.94970703125, + 27.86865234375, + 26.43994140625 + ], + "ended_at_unix": 1790740529.827453, + "bench_exit_code": 0 +} diff --git a/benchmarks/2026-09-29-recovery-wiring-rerun/sqlite-run.json b/benchmarks/2026-09-29-recovery-wiring-rerun/sqlite-run.json new file mode 100644 index 0000000..8c396bc --- /dev/null +++ b/benchmarks/2026-09-29-recovery-wiring-rerun/sqlite-run.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T03:53:29.311720+00:00", + "endpoint": "http://127.0.0.1:65298", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2116, + "errors": 0, + "first_error": null, + "ops_per_s": 423.14, + "items_per_s": 423.14, + "mean_ms": 2.36, + "p50_ms": 1.69, + "p95_ms": 6.2, + "p99_ms": 10.02 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3868, + "errors": 0, + "first_error": null, + "ops_per_s": 772.65, + "items_per_s": 772.65, + "mean_ms": 10.34, + "p50_ms": 9.17, + "p95_ms": 21.55, + "p99_ms": 29.46 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1745, + "errors": 0, + "first_error": null, + "ops_per_s": 348.79, + "items_per_s": 348.79, + "mean_ms": 2.87, + "p50_ms": 2.28, + "p95_ms": 6.85, + "p99_ms": 10.77 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3086, + "errors": 0, + "first_error": null, + "ops_per_s": 615.96, + "items_per_s": 615.96, + "mean_ms": 12.97, + "p50_ms": 11.16, + "p95_ms": 26.94, + "p99_ms": 39.52 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1932, + "errors": 0, + "first_error": null, + "ops_per_s": 386.09, + "items_per_s": 386.09, + "mean_ms": 2.59, + "p50_ms": 2.12, + "p95_ms": 6.05, + "p99_ms": 9.04 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3258, + "errors": 0, + "first_error": null, + "ops_per_s": 649.88, + "items_per_s": 649.88, + "mean_ms": 12.27, + "p50_ms": 10.46, + "p95_ms": 25.7, + "p99_ms": 38.59 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1269, + "errors": 0, + "first_error": null, + "ops_per_s": 253.78, + "items_per_s": 507.56, + "mean_ms": 3.94, + "p50_ms": 2.87, + "p95_ms": 10.14, + "p99_ms": 15.91 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.0, + "completed": 3209, + "errors": 0, + "first_error": null, + "ops_per_s": 641.22, + "items_per_s": 1282.45, + "mean_ms": 12.46, + "p50_ms": 10.79, + "p95_ms": 26.85, + "p99_ms": 36.99 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.01, + "completed": 916, + "errors": 0, + "first_error": null, + "ops_per_s": 182.98, + "items_per_s": 365.96, + "mean_ms": 5.46, + "p50_ms": 4.0, + "p95_ms": 14.52, + "p99_ms": 24.41 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2709, + "errors": 0, + "first_error": null, + "ops_per_s": 541.11, + "items_per_s": 1082.21, + "mean_ms": 14.76, + "p50_ms": 12.62, + "p95_ms": 32.02, + "p99_ms": 43.04 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1911, + "errors": 0, + "first_error": null, + "ops_per_s": 381.88, + "items_per_s": 0.0, + "mean_ms": 2.62, + "p50_ms": 2.06, + "p95_ms": 6.22, + "p99_ms": 9.88 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2793, + "errors": 0, + "first_error": null, + "ops_per_s": 557.12, + "items_per_s": 0.0, + "mean_ms": 14.32, + "p50_ms": 12.75, + "p95_ms": 28.81, + "p99_ms": 42.39 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1528, + "errors": 0, + "first_error": null, + "ops_per_s": 305.57, + "items_per_s": 0.0, + "mean_ms": 3.27, + "p50_ms": 2.63, + "p95_ms": 7.48, + "p99_ms": 11.06 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.0, + "completed": 4305, + "errors": 0, + "first_error": null, + "ops_per_s": 860.62, + "items_per_s": 0.0, + "mean_ms": 9.29, + "p50_ms": 8.18, + "p95_ms": 19.48, + "p99_ms": 27.19 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1526, + "errors": 0, + "first_error": null, + "ops_per_s": 305.12, + "items_per_s": 305.12, + "mean_ms": 3.28, + "p50_ms": 2.4, + "p95_ms": 8.34, + "p99_ms": 13.98 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.02, + "completed": 1594, + "errors": 0, + "first_error": null, + "ops_per_s": 317.73, + "items_per_s": 317.73, + "mean_ms": 25.13, + "p50_ms": 20.62, + "p95_ms": 58.22, + "p99_ms": 77.07 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.01, + "completed": 923, + "errors": 0, + "first_error": null, + "ops_per_s": 184.41, + "items_per_s": 184.41, + "mean_ms": 5.42, + "p50_ms": 4.11, + "p95_ms": 13.5, + "p99_ms": 19.11 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.04, + "completed": 1758, + "errors": 0, + "first_error": null, + "ops_per_s": 349.01, + "items_per_s": 349.01, + "mean_ms": 22.81, + "p50_ms": 19.0, + "p95_ms": 50.13, + "p99_ms": 73.49 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1049, + "errors": 0, + "first_error": null, + "ops_per_s": 209.71, + "items_per_s": 209.71, + "mean_ms": 4.77, + "p50_ms": 3.44, + "p95_ms": 11.91, + "p99_ms": 19.74 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 1725, + "errors": 0, + "first_error": null, + "ops_per_s": 344.37, + "items_per_s": 344.37, + "mean_ms": 23.2, + "p50_ms": 18.28, + "p95_ms": 55.85, + "p99_ms": 73.68 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.02, + "completed": 486, + "errors": 0, + "first_error": null, + "ops_per_s": 96.72, + "items_per_s": 193.45, + "mean_ms": 10.33, + "p50_ms": 8.2, + "p95_ms": 24.59, + "p99_ms": 37.93 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.07, + "completed": 807, + "errors": 0, + "first_error": null, + "ops_per_s": 159.27, + "items_per_s": 318.54, + "mean_ms": 49.85, + "p50_ms": 43.82, + "p95_ms": 107.01, + "p99_ms": 132.95 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 340, + "errors": 0, + "first_error": null, + "ops_per_s": 67.94, + "items_per_s": 135.89, + "mean_ms": 14.71, + "p50_ms": 11.97, + "p95_ms": 33.46, + "p99_ms": 50.23 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.02, + "completed": 1085, + "errors": 0, + "first_error": null, + "ops_per_s": 215.97, + "items_per_s": 431.94, + "mean_ms": 36.92, + "p50_ms": 30.88, + "p95_ms": 78.71, + "p99_ms": 103.8 + } + ] +} diff --git a/benchmarks/2026-09-29-rustfs-put-probe/README.md b/benchmarks/2026-09-29-rustfs-put-probe/README.md new file mode 100644 index 0000000..61aa995 --- /dev/null +++ b/benchmarks/2026-09-29-rustfs-put-probe/README.md @@ -0,0 +1,31 @@ +# Direct RustFS PUT diagnostic + +This probe isolates the object store from BeyondDB's DynamoDB request path. +It sent signed S3 `PutObject` calls for unique 1 KiB keys to a fresh container +pinned to `ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. +Each client used its own boto3 S3 client, disabled SDK retries, and ran a +five-second closed loop. All calls succeeded. + +| Clients | Completed PUTs | PUTs/s | p50 | p95 | p99 | +| ---: | ---: | ---: | ---: | ---: | ---: | +| 1 | 117 | 23.4 | 31.53 ms | 89.45 ms | 104.34 ms | +| 8 | 719 | 142.8 | 51.56 ms | 109.95 ms | 128.77 ms | +| 32 | 1,579 | 308.9 | 88.59 ms | 190.72 ms | 246.70 ms | + +The one-minute load average climbed from 38.7 to 43.3 on the 12-logical-CPU +host while other virtual machines were active. These numbers are **not** an +idle-host RustFS capacity measurement or a bound on BeyondDB throughput. +BeyondDB's Cell publication includes more work than one S3 PUT. The probe +does show that object-store timing is material under the same host conditions +as the recent BeyondDB release tests; tuning only item SQL cannot establish +SQLite write parity on this fixture. + +The pinned Cellule revision exposes a node-log durability supervisor and +follower fsync proofs. BeyondDB currently does not enroll followers or provide +the authenticated node-log transport and authority integration required to +use that path. A production comparison needs a multi-node fixture, verified +recovery after owner loss, and sustained mixed workloads before claiming that +follower durability improves the write gap. + +See [raw results](results.json). This probe used a fresh bucket, so it does +not include any BeyondDB table or item operations. diff --git a/benchmarks/2026-09-29-rustfs-put-probe/results.json b/benchmarks/2026-09-29-rustfs-put-probe/results.json new file mode 100644 index 0000000..72c699b --- /dev/null +++ b/benchmarks/2026-09-29-rustfs-put-probe/results.json @@ -0,0 +1,35 @@ +[ + { + "clients": 1, + "completed": 117, + "errors": 0, + "elapsed_s": 5.0, + "req_s": 23.4, + "p50_ms": 31.53, + "p95_ms": 89.45, + "p99_ms": 104.34, + "host_load": [38.677734375, 25.228515625, 20.44775390625] + }, + { + "clients": 8, + "completed": 719, + "errors": 0, + "elapsed_s": 5.04, + "req_s": 142.8, + "p50_ms": 51.56, + "p95_ms": 109.95, + "p99_ms": 128.77, + "host_load": [42.466796875, 26.2373046875, 20.83154296875] + }, + { + "clients": 32, + "completed": 1579, + "errors": 0, + "elapsed_s": 5.11, + "req_s": 308.9, + "p50_ms": 88.59, + "p95_ms": 190.72, + "p99_ms": 246.7, + "host_load": [43.31005859375, 26.68115234375, 21.02001953125] + } +] diff --git a/benchmarks/2026-09-29-sqlite-comparison/README.md b/benchmarks/2026-09-29-sqlite-comparison/README.md new file mode 100644 index 0000000..eae5c05 --- /dev/null +++ b/benchmarks/2026-09-29-sqlite-comparison/README.md @@ -0,0 +1,51 @@ +# Release comparison: BeyondDB and ExtendDB SQLite + +On September 29, 2026 (Pacific time), the same Mac14,13 host ran two fresh, +sequential, five-second-per-case fixtures. Both used `scripts/bench.py`, signed +boto3, no SDK retries, 64 seeded items with a 1 KiB payload attribute, and +one/eight closed-loop clients. Batch and transaction operations used two items +per request. All 24 cases on each backend completed with zero SDK request +errors. + +- ExtendDB revision `7eaa89b437feed0af0f05883d3f1493f86c6fc6d` ran its + release `sqlite,dev-mode` binary with a fresh file-backed SQLite database. + Dev mode verifies SigV4 but uses open authorization. +- BeyondDB revision `af8fab7` ran its release binary with four initial + partitions, 12 SQL workers, 128 active Cell slots, a 1 GiB retained-byte + budget, `auth_cache_enabled: true`, and a fresh RustFS container pinned to + `ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. + BeyondDB verified IAM as well as SigV4. +- Host load was high and changed between fixtures: the one-minute average was + 14.5 at SQLite startup, 21.8 at SQLite completion, 23.7 at BeyondDB startup, + and 28.8 at BeyondDB completion on 12 logical CPUs. Other virtual machines + were active. These are contemporary diagnostics, not a controlled capacity + or causal comparison. + +| API | SQLite requests/s, 1 / 8 clients | BeyondDB requests/s, 1 / 8 clients | +| --- | ---: | ---: | +| GetItem | 645.75 / 1,015.78 | 582.99 / 880.55 | +| Query | 662.96 / 841.20 | 538.11 / 536.53 | +| Scan | 572.81 / 737.67 | 281.28 / 787.87 | +| BatchGetItem | 352.52 / 903.73 | 375.54 / 631.17 | +| TransactGetItems | 572.18 / 937.65 | 1.50 / 4.76 | +| DescribeTable | 984.46 / 1,119.81 | 285.22 / 384.20 | +| ListTables | 1,089.09 / 1,357.85 | 214.85 / 472.34 | +| PutItem | 810.72 / 1,302.96 | 9.09 / 33.24 | +| UpdateItem | 734.55 / 1,272.89 | 10.40 / 28.57 | +| DeleteItem (missing) | 871.35 / 444.58 | 17.92 / 72.85 | +| BatchWriteItem | 75.96 / 166.93 | 9.82 / 35.11 | +| TransactWriteItems | 78.62 / 285.20 | 1.72 / 3.34 | + +BeyondDB exceeded SQLite only for eight-client `Scan` and one-client +`BatchGetItem` in this sample. It stayed close on `GetItem`, but durable +mutations and cross-partition transactions remained far slower. Its server +logged eight deferred background operations, including Cell mailbox-byte +exhaustion and incomplete transaction recovery. Zero SDK request errors do +not establish sustainable capacity. + +See the raw [SQLite results](extenddb-sqlite.json), [BeyondDB results](beyonddb-rustfs.json), +and [BeyondDB warnings](beyonddb-warnings.txt). Repeat on an otherwise idle +host with equal authorization policy and longer runs before drawing production +sizing conclusions. The two backends do not have equal durability contracts: +BeyondDB awaits published Cell state in RustFS for each durable write, while +ExtendDB SQLite writes to its local database file. diff --git a/benchmarks/2026-09-29-sqlite-comparison/beyonddb-rustfs.json b/benchmarks/2026-09-29-sqlite-comparison/beyonddb-rustfs.json new file mode 100644 index 0000000..8a71646 --- /dev/null +++ b/benchmarks/2026-09-29-sqlite-comparison/beyonddb-rustfs.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T00:08:19.019994+00:00", + "endpoint": "http://127.0.0.1:57586", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2917, + "errors": 0, + "first_error": null, + "ops_per_s": 582.99, + "items_per_s": 582.99, + "mean_ms": 1.71, + "p50_ms": 1.32, + "p95_ms": 3.94, + "p99_ms": 7.11 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 4406, + "errors": 0, + "first_error": null, + "ops_per_s": 880.55, + "items_per_s": 880.55, + "mean_ms": 9.07, + "p50_ms": 7.93, + "p95_ms": 18.36, + "p99_ms": 26.6 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2692, + "errors": 0, + "first_error": null, + "ops_per_s": 538.11, + "items_per_s": 538.11, + "mean_ms": 1.86, + "p50_ms": 1.72, + "p95_ms": 3.25, + "p99_ms": 5.85 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2687, + "errors": 0, + "first_error": null, + "ops_per_s": 536.53, + "items_per_s": 536.53, + "mean_ms": 14.88, + "p50_ms": 12.93, + "p95_ms": 31.32, + "p99_ms": 46.03 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1407, + "errors": 0, + "first_error": null, + "ops_per_s": 281.28, + "items_per_s": 281.28, + "mean_ms": 3.55, + "p50_ms": 2.96, + "p95_ms": 6.88, + "p99_ms": 14.1 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 3943, + "errors": 0, + "first_error": null, + "ops_per_s": 787.87, + "items_per_s": 787.87, + "mean_ms": 10.14, + "p50_ms": 8.86, + "p95_ms": 18.96, + "p99_ms": 29.54 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1878, + "errors": 0, + "first_error": null, + "ops_per_s": 375.54, + "items_per_s": 751.09, + "mean_ms": 2.66, + "p50_ms": 2.35, + "p95_ms": 4.17, + "p99_ms": 10.62 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.01, + "completed": 3163, + "errors": 0, + "first_error": null, + "ops_per_s": 631.17, + "items_per_s": 1262.34, + "mean_ms": 12.66, + "p50_ms": 9.91, + "p95_ms": 28.41, + "p99_ms": 58.37 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.35, + "completed": 8, + "errors": 0, + "first_error": null, + "ops_per_s": 1.5, + "items_per_s": 2.99, + "mean_ms": 668.72, + "p50_ms": 686.4, + "p95_ms": 903.64, + "p99_ms": 903.64 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 6.52, + "completed": 31, + "errors": 0, + "first_error": null, + "ops_per_s": 4.76, + "items_per_s": 9.51, + "mean_ms": 1559.58, + "p50_ms": 1788.55, + "p95_ms": 2419.91, + "p99_ms": 2482.79 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.01, + "completed": 1428, + "errors": 0, + "first_error": null, + "ops_per_s": 285.22, + "items_per_s": 0.0, + "mean_ms": 3.5, + "p50_ms": 2.42, + "p95_ms": 6.61, + "p99_ms": 27.94 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.03, + "completed": 1934, + "errors": 0, + "first_error": null, + "ops_per_s": 384.2, + "items_per_s": 0.0, + "mean_ms": 20.76, + "p50_ms": 15.65, + "p95_ms": 55.3, + "p99_ms": 80.13 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.01, + "completed": 1077, + "errors": 0, + "first_error": null, + "ops_per_s": 214.85, + "items_per_s": 0.0, + "mean_ms": 4.65, + "p50_ms": 2.42, + "p95_ms": 14.64, + "p99_ms": 29.65 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2365, + "errors": 0, + "first_error": null, + "ops_per_s": 472.34, + "items_per_s": 0.0, + "mean_ms": 16.9, + "p50_ms": 12.52, + "p95_ms": 40.78, + "p99_ms": 72.84 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.06, + "completed": 46, + "errors": 0, + "first_error": null, + "ops_per_s": 9.09, + "items_per_s": 9.09, + "mean_ms": 109.98, + "p50_ms": 69.47, + "p95_ms": 307.41, + "p99_ms": 349.17 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.2, + "completed": 173, + "errors": 0, + "first_error": null, + "ops_per_s": 33.24, + "items_per_s": 33.24, + "mean_ms": 235.59, + "p50_ms": 162.47, + "p95_ms": 635.55, + "p99_ms": 1181.64 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 52, + "errors": 0, + "first_error": null, + "ops_per_s": 10.4, + "items_per_s": 10.4, + "mean_ms": 96.18, + "p50_ms": 57.81, + "p95_ms": 213.54, + "p99_ms": 593.38 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.46, + "completed": 156, + "errors": 0, + "first_error": null, + "ops_per_s": 28.57, + "items_per_s": 28.57, + "mean_ms": 267.31, + "p50_ms": 237.02, + "p95_ms": 636.51, + "p99_ms": 709.17 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.02, + "completed": 90, + "errors": 0, + "first_error": null, + "ops_per_s": 17.92, + "items_per_s": 17.92, + "mean_ms": 55.81, + "p50_ms": 44.85, + "p95_ms": 115.46, + "p99_ms": 154.88 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.11, + "completed": 372, + "errors": 0, + "first_error": null, + "ops_per_s": 72.85, + "items_per_s": 72.85, + "mean_ms": 108.88, + "p50_ms": 86.11, + "p95_ms": 260.5, + "p99_ms": 413.73 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.09, + "completed": 50, + "errors": 0, + "first_error": null, + "ops_per_s": 9.82, + "items_per_s": 19.65, + "mean_ms": 101.79, + "p50_ms": 68.37, + "p95_ms": 227.47, + "p99_ms": 351.32 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.33, + "completed": 187, + "errors": 0, + "first_error": null, + "ops_per_s": 35.11, + "items_per_s": 70.23, + "mean_ms": 222.13, + "p50_ms": 190.26, + "p95_ms": 429.6, + "p99_ms": 857.9 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.81, + "completed": 10, + "errors": 0, + "first_error": null, + "ops_per_s": 1.72, + "items_per_s": 3.44, + "mean_ms": 580.73, + "p50_ms": 506.5, + "p95_ms": 712.4, + "p99_ms": 712.4 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 6.29, + "completed": 21, + "errors": 0, + "first_error": null, + "ops_per_s": 3.34, + "items_per_s": 6.68, + "mean_ms": 2206.93, + "p50_ms": 1971.16, + "p95_ms": 3720.25, + "p99_ms": 3720.25 + } + ] +} diff --git a/benchmarks/2026-09-29-sqlite-comparison/beyonddb-warnings.txt b/benchmarks/2026-09-29-sqlite-comparison/beyonddb-warnings.txt new file mode 100644 index 0000000..d67c630 --- /dev/null +++ b/benchmarks/2026-09-29-sqlite-comparison/beyonddb-warnings.txt @@ -0,0 +1,8 @@ +2026-09-30T00:08:35.927126Z WARN beyonddb::provision::capacity: capacity sweep deferred account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T00:08:57.597199Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T00:09:08.892369Z WARN beyonddb::provision::capacity: capacity sweep deferred account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T00:09:10.525741Z WARN beyonddb::provision::transactions: transaction recovery deferred error=cross-Cell participant resolution is incomplete +2026-09-30T00:09:14.980612Z WARN beyonddb: catalog stream retention sweep failed account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T00:09:17.699118Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T00:09:19.892612Z WARN beyonddb::backend::global_index: global index projection deferred error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes +2026-09-30T00:09:20.917341Z WARN beyonddb::provision::capacity: capacity sweep deferred account_id="123456789012" error=Cell invocation did not start: Cell runtime capacity exhausted: Cell mailbox bytes diff --git a/benchmarks/2026-09-29-sqlite-comparison/extenddb-sqlite.json b/benchmarks/2026-09-29-sqlite-comparison/extenddb-sqlite.json new file mode 100644 index 0000000..f0a74e9 --- /dev/null +++ b/benchmarks/2026-09-29-sqlite-comparison/extenddb-sqlite.json @@ -0,0 +1,387 @@ +{ + "started_at": "2026-09-30T00:05:39.070066+00:00", + "endpoint": "http://127.0.0.1:56035", + "table": "PerfData", + "seconds_per_case": 5, + "payload_bytes": 1024, + "seed_keys": 64, + "sort_key": null, + "strong_reads": true, + "sdk_retries": 0, + "operations": [ + "get", + "query", + "scan", + "batch_get", + "transact_get", + "describe_table", + "list_tables", + "put", + "update", + "delete_missing", + "batch_write", + "transact_write" + ], + "cases": [ + { + "operation": "get", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 3229, + "errors": 0, + "first_error": null, + "ops_per_s": 645.75, + "items_per_s": 645.75, + "mean_ms": 1.55, + "p50_ms": 1.45, + "p95_ms": 2.44, + "p99_ms": 3.2 + }, + { + "operation": "get", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 5082, + "errors": 0, + "first_error": null, + "ops_per_s": 1015.78, + "items_per_s": 1015.78, + "mean_ms": 7.87, + "p50_ms": 7.36, + "p95_ms": 14.15, + "p99_ms": 17.97 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 3316, + "errors": 0, + "first_error": null, + "ops_per_s": 662.96, + "items_per_s": 662.96, + "mean_ms": 1.51, + "p50_ms": 1.35, + "p95_ms": 2.53, + "p99_ms": 4.1 + }, + { + "operation": "query", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 4210, + "errors": 0, + "first_error": null, + "ops_per_s": 841.2, + "items_per_s": 841.2, + "mean_ms": 9.5, + "p50_ms": 8.59, + "p95_ms": 18.7, + "p99_ms": 25.89 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2866, + "errors": 0, + "first_error": null, + "ops_per_s": 572.81, + "items_per_s": 572.81, + "mean_ms": 1.74, + "p50_ms": 1.61, + "p95_ms": 3.07, + "p99_ms": 4.53 + }, + { + "operation": "scan", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 3691, + "errors": 0, + "first_error": null, + "ops_per_s": 737.67, + "items_per_s": 737.67, + "mean_ms": 10.83, + "p50_ms": 9.55, + "p95_ms": 22.31, + "p99_ms": 30.8 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 1763, + "errors": 0, + "first_error": null, + "ops_per_s": 352.52, + "items_per_s": 705.05, + "mean_ms": 2.83, + "p50_ms": 2.37, + "p95_ms": 5.63, + "p99_ms": 10.86 + }, + { + "operation": "batch_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.0, + "completed": 4522, + "errors": 0, + "first_error": null, + "ops_per_s": 903.73, + "items_per_s": 1807.46, + "mean_ms": 8.84, + "p50_ms": 8.07, + "p95_ms": 16.64, + "p99_ms": 22.48 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 2862, + "errors": 0, + "first_error": null, + "ops_per_s": 572.18, + "items_per_s": 1144.36, + "mean_ms": 1.75, + "p50_ms": 1.64, + "p95_ms": 2.54, + "p99_ms": 3.42 + }, + { + "operation": "transact_get", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.0, + "completed": 4691, + "errors": 0, + "first_error": null, + "ops_per_s": 937.65, + "items_per_s": 1875.3, + "mean_ms": 8.53, + "p50_ms": 7.74, + "p95_ms": 16.29, + "p99_ms": 21.64 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 4923, + "errors": 0, + "first_error": null, + "ops_per_s": 984.46, + "items_per_s": 0.0, + "mean_ms": 1.02, + "p50_ms": 0.98, + "p95_ms": 1.48, + "p99_ms": 1.95 + }, + { + "operation": "describe_table", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.0, + "completed": 5601, + "errors": 0, + "first_error": null, + "ops_per_s": 1119.81, + "items_per_s": 0.0, + "mean_ms": 7.14, + "p50_ms": 6.57, + "p95_ms": 13.05, + "p99_ms": 16.92 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 1, + "elapsed_s": 5.0, + "completed": 5447, + "errors": 0, + "first_error": null, + "ops_per_s": 1089.09, + "items_per_s": 0.0, + "mean_ms": 0.92, + "p50_ms": 0.8, + "p95_ms": 1.65, + "p99_ms": 2.49 + }, + { + "operation": "list_tables", + "items_per_request": 0, + "clients": 8, + "elapsed_s": 5.0, + "completed": 6792, + "errors": 0, + "first_error": null, + "ops_per_s": 1357.85, + "items_per_s": 0.0, + "mean_ms": 5.89, + "p50_ms": 5.43, + "p95_ms": 10.51, + "p99_ms": 14.14 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 4054, + "errors": 0, + "first_error": null, + "ops_per_s": 810.72, + "items_per_s": 810.72, + "mean_ms": 1.23, + "p50_ms": 1.15, + "p95_ms": 1.91, + "p99_ms": 2.99 + }, + { + "operation": "put", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 6518, + "errors": 0, + "first_error": null, + "ops_per_s": 1302.96, + "items_per_s": 1302.96, + "mean_ms": 6.14, + "p50_ms": 5.76, + "p95_ms": 10.73, + "p99_ms": 14.05 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 3675, + "errors": 0, + "first_error": null, + "ops_per_s": 734.55, + "items_per_s": 734.55, + "mean_ms": 1.36, + "p50_ms": 1.27, + "p95_ms": 2.15, + "p99_ms": 3.07 + }, + { + "operation": "update", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.0, + "completed": 6367, + "errors": 0, + "first_error": null, + "ops_per_s": 1272.89, + "items_per_s": 1272.89, + "mean_ms": 6.28, + "p50_ms": 5.84, + "p95_ms": 11.02, + "p99_ms": 14.76 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 1, + "elapsed_s": 5.0, + "completed": 4357, + "errors": 0, + "first_error": null, + "ops_per_s": 871.35, + "items_per_s": 871.35, + "mean_ms": 1.15, + "p50_ms": 1.09, + "p95_ms": 1.77, + "p99_ms": 2.31 + }, + { + "operation": "delete_missing", + "items_per_request": 1, + "clients": 8, + "elapsed_s": 5.01, + "completed": 2229, + "errors": 0, + "first_error": null, + "ops_per_s": 444.58, + "items_per_s": 444.58, + "mean_ms": 17.96, + "p50_ms": 10.01, + "p95_ms": 63.14, + "p99_ms": 106.36 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.0, + "completed": 380, + "errors": 0, + "first_error": null, + "ops_per_s": 75.96, + "items_per_s": 151.92, + "mean_ms": 13.15, + "p50_ms": 7.91, + "p95_ms": 39.33, + "p99_ms": 66.07 + }, + { + "operation": "batch_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.07, + "completed": 847, + "errors": 0, + "first_error": null, + "ops_per_s": 166.93, + "items_per_s": 333.85, + "mean_ms": 47.66, + "p50_ms": 35.74, + "p95_ms": 124.11, + "p99_ms": 234.28 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 1, + "elapsed_s": 5.01, + "completed": 394, + "errors": 0, + "first_error": null, + "ops_per_s": 78.62, + "items_per_s": 157.23, + "mean_ms": 12.72, + "p50_ms": 7.8, + "p95_ms": 37.28, + "p99_ms": 87.75 + }, + { + "operation": "transact_write", + "items_per_request": 2, + "clients": 8, + "elapsed_s": 5.04, + "completed": 1438, + "errors": 0, + "first_error": null, + "ops_per_s": 285.2, + "items_per_s": 570.4, + "mean_ms": 28.0, + "p50_ms": 18.22, + "p95_ms": 85.83, + "p99_ms": 130.56 + } + ] +} diff --git a/docs/cross-cell-transactions.md b/docs/cross-cell-transactions.md index 4791209..4179dbc 100644 --- a/docs/cross-cell-transactions.md +++ b/docs/cross-cell-transactions.md @@ -188,11 +188,14 @@ published. Token lookup precedes current table routing and uses that same coordinator authority. The earlier account claims and local token receipts have been removed. -`TransactGetItems` uses the same durable coordinator for every request, with -shared key locks and immutable participant images. Saved images are fetched -individually, including when all requested keys share one Cell. Account and data reads reject unresolved -write intents instead of returning images that could predate a published -commit. Full DynamoDB compatibility and fleet-scale qualification remain open. +`TransactGetItems` uses one read-only Cell query when every key routes to the +same Cell and its encoded response fits the Cell wire limit. The Cell worker +checks locks and reads the items in one serialized callback. Cross-Cell reads, +and same-Cell reads whose encoded response exceeds that limit, use the durable +coordinator with shared locks and individually fetched saved images. Account +and data reads reject unresolved write intents instead of returning images +that could predate a published commit. Full DynamoDB compatibility and +fleet-scale qualification remain open. Each coordinator now indexes records with unresolved participants and exposes bounded cursor pages. A new owner can discover both undecided and decided @@ -305,9 +308,10 @@ Within one transaction, prepare remains sequential and terminal resolution has a four-participant window. Coordinator progress writes remain serialized by their Cell owner. -All transactional reads pay the same phase cost and persist their captured -images; assembly additionally queries each saved item. A same-Cell read therefore -requires six phase commands, at least two upload commands, and result queries. Before optimizing the protocol, measure publication latency, +Cross-Cell reads and oversized same-Cell reads pay the coordinator phase cost +and persist their captured images; assembly additionally queries each saved +item. Same-Cell responses that fit the wire limit avoid this publication work. +Before optimizing the coordinator protocol, measure publication latency, participant count, hot-key conflicts, recovery competition, and retained bytes. Terminal resolution now overlaps independent participants after a durable @@ -448,39 +452,44 @@ codec framing. BeyondDB's `Json` serializes ExtendDB attribute values as DynamoDB JSON. Binary values become base64; control characters can expand to six JSON bytes per source byte. SQL chunking does not change that outer limit. -The former same-Cell read optimization returned every requested image in one -query. A host reproduction stored ten 380-KiB binary values successfully, then -failed `TransactGetItems` with `Cell wire codec failed`. The aggregate raw +The same-Cell read query returns every requested image in one response. A host +reproduction stored ten 380-KiB binary values successfully, then failed +`TransactGetItems` with `Cell wire codec failed`. The aggregate raw payload was 3,891,200 bytes, below 4 MiB; its base64 alone was 5,188,280 bytes. The signed SDK reproduction against a remote data owner returned HTTP 503. Neither failure demonstrates partial writes: both occurred during a read. -All transactional reads now use shared prepare, durable decision, resolution, -and individual saved-image retrieval. The account and data aggregate snapshot -queries and the adapter's route-dependent branch are removed. The public API -retains one ordered, serializable result; HTTP response assembly happens above -the Cell wire boundary. +If the Cell query exceeds its wire limit, the adapter retries through the +durable coordinator read protocol. That protocol uses shared prepare, durable +decision, resolution, and individual saved-image retrieval. A read-only query +never commits a mutation, so this fallback does not duplicate work. The public +API retains one ordered, serializable result; HTTP response assembly happens +above the Cell wire boundary. A saved image remains immutable after lock release, so fetching its siblings later cannot mix newer live versions into the result. The adapter and participant still enforce the raw 4-MiB read limit. +The adapter fetches saved images sequentially from each Cell to stay within +that Cell's fixed mailbox budget; different participant Cells can be read in +parallel. -**Is this the best fix here?** Reusing the existing prepare/resolve and saved-image -lifecycle removes the failing path without adding a separate snapshot retention -protocol. It deliberately trades the former optimization for one recovery path. +This fallback reuses the existing prepare/resolve and saved-image lifecycle for +large responses without adding a separate snapshot retention protocol. Smaller +same-Cell reads retain the faster read-only query. -**Cost:** same-Cell reads publish six phase commands plus input uploads and retain recovery/image -records. They depend on coordinator availability and contend with writes during -prepare. This is a correctness tradeoff, not a read performance optimization. -A future fast path needs a bounded snapshot handle and explicit retention and -recovery rules; repeatedly reading live pages would violate the transaction. +**Cost:** oversized same-Cell reads publish coordinator phase commands and retain +recovery/image records. They depend on coordinator availability and contend with +writes during prepare. A future large-read fast path needs a bounded snapshot +handle and explicit retention and recovery rules; repeatedly reading live pages +would violate the transaction. `tests/account_cell.rs` covers ten binary images on an account participant. The shared signed SDK fixture covers ten 380-KiB binary values and four 380-KiB control-character strings, each group deliberately routed to one data -Cell. It checks every byte in reversed request order and repeats reads after -owner replacement and hard process restart. Existing conflict, absent-image, -projection, and cross-Cell tests exercise the same read protocol. +Cell. It checks every byte in reversed request order. The broader restart +fixture also exercises these reads, though a separate owner-activation failure +currently prevents the full restart suite from passing. Existing conflict, +absent-image, projection, and cross-Cell tests exercise the same read protocol. The expanded `scripts/probe-transaction-size.py` also ran both read cases against the verified DynamoDB Local 3.3.1 reference: both succeeded, all diff --git a/docs/deployment.md b/docs/deployment.md index 89993c3..1168d3f 100644 --- a/docs/deployment.md +++ b/docs/deployment.md @@ -28,8 +28,12 @@ Prepare these resources before starting the server: - A 32-byte binary encryption key file. Losing it makes stored access-key secrets unreadable. It is shared by nodes in one fleet. - A writable scratch directory and enough RAM and disk for the configured - Cell and capture budgets. `disk_budget_bytes` and - `split_threshold_bytes` must be positive. + Cell and capture budgets. `disk_budget_bytes`, `node_retained_bytes`, and + `split_threshold_bytes` must be positive. `node_retained_bytes` bounds + in-flight Cell command and publication state; it defaults to 1 GiB so a + write-heavy node does not hit the runtime mailbox ceiling before durable + publication catches up. Lower it on a memory-constrained node or raise it + for a larger workload after measuring memory use. ### Local object-store fixture @@ -121,6 +125,9 @@ Save the next JSON block as `config.json` in the repository root. Replace its ob "node_id": "01994f26-5966-7b20-8b58-2fddf198a321", "data_dir": "/srv/beyonddb/scratch", "disk_budget_bytes": 107374182400, + "node_retained_bytes": 1073741824, + "max_active_cells": 128, + "sql_workers": 12, "encryption_key_file": "/etc/beyonddb/encryption.key", "region": "us-east-1", "peer_bind": "127.0.0.1:9001", @@ -151,9 +158,24 @@ generations can be changed by another node. Set it to `true` only when the cache's 60-second cross-node visibility window is acceptable. Local management mutations invalidate cached credentials, policies, boundaries, and table metadata immediately; changes made through another node become visible after -the cache TTL. - -The parser rejects unknown fields. `initial_partitions` defaults to one and can provision 1โ€“256 initial data Cells per new table. The split threshold defaults to 256 MiB of occupied SQLite pages. `node_id` identifies a physical node; each running node needs a distinct ID and scratch path. +the cache TTL. Enabling the flag also caches complete routed directory pages +for point operations. The owning data Cell rejects a stale epoch and the +server drops that route entry, so a split is refreshed on the next request. It +also keeps immutable catalog proofs and resident local Cell handles for 500 ms; +the authority check resumes after that window, and a drained Cell handle still +rejects work immediately. This short owner cache improves warm local latency +while bounding visibility of an ownership change. + +The parser rejects unknown fields. `initial_partitions` defaults to one and can provision 1โ€“256 initial data Cells per new table. `max_active_cells` defaults to 128 and reserves the Cell runtime capacity for account, coordinator, management, and data Cells together; size it for the number of simultaneously resident Cells on the node and the available memory. `sql_workers` is optional; when omitted, the runtime derives the worker count from host parallelism, capped at sixteen. Set it explicitly when a node serves many partitions and you have measured enough CPU and memory headroom. Each worker owns its SQLite connections, so increasing the value does not make one hot Cell publish concurrently. The split threshold defaults to 256 MiB of occupied SQLite pages. `node_id` identifies a physical node; each running node needs a distinct ID and scratch path. + +`follower_store_bytes` is an optional positive disk budget for persistent +follower lanes under `data_dir/follower-store`. Setting it opens a private, +authenticated node-log receiver; it does **not** enable follower durability or +improve write latency yet. BeyondDB still waits for object-store publication +and advertises no follower capacity. Reserve this budget in addition to +`disk_budget_bytes`, and retain the follower directory across process restart. +The [follower durability guide](follower-durability.md) tracks the remaining +enrollment, lifecycle, and recovery work. | Credential or file | Used by | Keep across restart? | | --- | --- | --- | diff --git a/docs/follower-durability.md b/docs/follower-durability.md new file mode 100644 index 0000000..bf63e03 --- /dev/null +++ b/docs/follower-durability.md @@ -0,0 +1,123 @@ +# Follower durability for low-latency writes + +## Why this work is needed + +The current BeyondDB server acknowledges a mutation after its Cell state is +published to the configured object store. On the recent four-partition local +RustFS fixture, this path remained far below file-backed ExtendDB SQLite for +`PutItem`, `UpdateItem`, `BatchWriteItem`, and transactions. A separate +[direct RustFS PUT probe](../benchmarks/2026-09-29-rustfs-put-probe/README.md) +also observed material object-store latency under heavy host contention. Those +short runs do not establish an idle-host throughput limit, but removing small +amounts of item SQL cannot eliminate the object publication round trip. + +The pinned Cellule revision already exposes a node-log durability supervisor, +follower stores, and a durability gate. BeyondDB has a lease-bound authority +adapter, an opt-in inbound follower receiver, a pinned mTLS outbound +transport, and a host-provider enrollment adapter. The provider is not +installed in the serving binary, and owner recovery is incomplete. No +follower mode should be advertised or enabled until that integration is +proven. + +```text +Current serving path Proposed multi-node path + +AWS SDK request AWS SDK request + | | +Cell command + local capture Cell command + local capture + | | +object-store publication append to every enrolled follower + | | +durable object proof follower fsync receipts + | | +SDK success authoritative log activation/proof + | + SDK success + + Object tiering continues; recovery must + replay any acknowledged untiered frames. +``` + +Cellule's gate requires every member in the enrolled follower set to fsync a +ticket before a fleet proof. Its first fleet proof also requires an +authoritative activation of that exact log epoch. Object proof remains a +fallback. These are durability conditions, not optional performance hints. + +## Product integration boundary + +| BeyondDB component | Required behavior | +| --- | --- | +| Follower store | Open `FollowerStore` in each node's durable data directory, reserve disk, and retain lanes across process restart. Advertise follower capacity only after the store and authenticated listener are ready. | +| Peer transport | Implement Cellule's `NodeLogTransport` append, seal, retire, and bounded tail operations over the private mTLS listener. Pin each remote certificate to its live node advertisement. Bound request bytes, time, and concurrent work. | +| Follower authorization | Match the mTLS identity to the advertised session. Use `NodeDirectory::authorize_log_append`, `authorize_log_retire`, and `authorize_log_recovery` before touching a lane. Reject wrong members, epochs, coverage watermarks, and unfenced recovery attempts. | +| Enrollment and authority | Implement `NodeDurabilityProvider` using `NodeDirectory::try_recruit_log`. Implement `NodeLogAuthority` with the directory's activate, coverage, and close CAS operations; reconcile CAS races with lease heartbeats without losing the enrolled log. Supply the exact session, node ID, members, lease guard, transport, and limits to `NodeDurabilityConfig`. | +| Recovery | Before public readiness after an owner loss, seal and fetch the failed owner's authorized follower tail, reconcile object coverage, and restore acknowledged Cell commits. Do not return success for a write whose proof cannot be recovered on a successor. | +| Lifecycle | Rotate and retire only after the recorded coverage barrier. Drain the node, settle publications, and preserve follower files if withdrawal or recovery has not completed. | + +The private `BeyonddbPeers` router now has a bounded node-log receiver when +`follower_store_bytes` is configured. It opens Cellule's persistent +`FollowerStore` beneath `data_dir`, requires a live mTLS identity bound to the +advertised session, and checks the directory's append, retirement, or fenced +recovery authority before touching a lane. The outbound +`PeerNodeLogTransport` resolves a live advertised member, pins its certificate +and key, and bounds requests, replies, and tail paging. Tests cover a durable +append and duplicate append over real two-identity mTLS, follower-store +reopen, wrong certificate, wrong epoch, and a seal attempt before owner +fencing. The recovery claimant test also fences the expired leader, +seals its own follower lane under directory authority, and passes that lane +through Cellule's bounded witness reader. A separate three-identity mTLS test +lets another claimant fence the leader, seal a remote follower, and read its +persisted tail through the bounded witness reader; sealing before the claim +is rejected. The test also confirms that a recovery claimant must advertise +Cellule's node-log protocol even when it offers no follower bytes. With an +opt-in persistent store, the serving binary now advertises that protocol but +zero follower bytes, so it can claim recovery without being recruited for +write acknowledgments. Retirement and successor Cell overlay attachment +still need end-to-end tests. + +The lease-bound `PublishedNodeLogAuthority` adapter can enroll a follower set +and apply the directory's activation, coverage, and close transitions. It +serializes those mutations with heartbeat refreshes and reloads the exact +session after an ambiguous CAS. `PeerNodeDurabilityProvider` gives Cellule's +host supervisor the enrolled members, authority, transport, and lease for +each epoch, but the serving binary does not install it while successor +recovery is unfinished. It advertises no usable follower capacity. Adding a +local in-process follower under a second logical node ID would not provide +an independent failure domain and must not be used as a production durability +shortcut. + +`recover_fenced_node_log` now composes Cellule's bounded witness reader, +tenant catalog inventory, overlay manifest pinning, and final directory seal +for a caller that already holds a fenced recovery claim. It resolves every +authenticated Cell scope through a durable tenant catalog and rejects missing, +ambiguous, or over-limit inventory before attaching an overlay. When normal +Cell takeover encounters an active, untiered node log, the opt-in serving path +claims fenced recovery, runs this coordinator, and only then passes its +takeover proof to Cellule. A product test now captures a real account Cell +frame, appends it to a persistent follower lane, and verifies that a +successor restores the untiered commit before takeover. This path still +lacks a multi-node crash-after-acknowledged-write SDK test; the provider +remains uninstalled and follower-backed acknowledgments remain disabled. + +## Verification before comparing throughput + +1. Prove append, duplicate append, lost acknowledgement, seal, tail paging, + retirement, wrong epoch, and forged peer rejection against persisted + follower lanes. +2. Run at least three BeyondDB nodes with separate data directories and + authenticated private endpoints. Kill an owner after an acknowledged + signed SDK write and verify the item, stream intent, and transaction result + through a replacement owner before reporting that API as supported in + follower mode. +3. Repeat the full signed boto3 API fixture on a release build with raw JSON, + errors, p50/p95/p99, host and object-store load, node count, follower proof + source, and background-work logs. Run sustained mixed traffic and hot-key + cases as well as short closed-loop samples. +4. Compare every API and client count with a fresh file-backed ExtendDB SQLite + fixture. Report the differing IAM and durability contracts. A local + three-process fixture establishes functionality, not fleet-scale capacity + or independent-host fault tolerance. + +The all-API SQLite performance objective remains open. Follower durability is +the principal architectural path to test for durable write latency; its actual +gain must be measured after the complete recovery contract works. diff --git a/docs/implementation-status.md b/docs/implementation-status.md index ffe74af..04027cc 100644 --- a/docs/implementation-status.md +++ b/docs/implementation-status.md @@ -106,6 +106,8 @@ file contains exactly 32 raw bytes and must be retained across restarts. "node_id": "01994f26-5966-7b20-8b58-2fddf198a321", "data_dir": "/srv/beyonddb/scratch", "disk_budget_bytes": 107374182400, + "max_active_cells": 128, + "sql_workers": 12, "encryption_key_file": "/etc/beyonddb/encryption.key", "region": "us-east-1", "peer_bind": "0.0.0.0:9001", diff --git a/docs/performance.md b/docs/performance.md index 2bc2357..94551ef 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -1,6 +1,56 @@ # Measure BeyondDB performance -BeyondDB does not yet have a qualified production throughput or latency target. The September 2026 single-node release comparison below shows a large gap from ExtendDB's file-backed SQLite backend. Do not use these numbers to plan a fleet. BeyondDB's request path and durability contract differ: an item write waits for Cellule to publish the committed Cell state to an object store. +BeyondDB does not yet have a qualified production throughput or latency target. Earlier September 2026 single-node samples suggested that warm point reads could exceed the file-backed SQLite fixture, while durable writes and transactions remained slower. The refresh below did not reproduce those high read rates. Do not use these numbers to plan a fleet. BeyondDB's request path and durability contract differ: an item write waits for Cellule to publish the committed Cell state to an object store. + +The latest full signed-API rerun pinned Cellule to `9e17746` and ExtendDB +SQLite to `7eaa89b`. Both completed all 24 five-second cases without +foreground SDK errors. At eight clients, BeyondDB/SQLite measured 235/507 +`GetItem`, 78/55 `PutItem`, and 1.50/131 `TransactWriteItems` requests/s. +BeyondDB exceeded SQLite in BatchGetItem and metadata reads at both client +counts, and in PutItem at eight clients, but did not meet the all-API target. +It logged deferred background work. External host load changed materially +between fixtures. See the [full table, p95 latencies, fixture, and raw +JSON](../benchmarks/2026-09-29-cellule-main-rerun/README.md); the result does +not establish production capacity or a controlled speed ratio. + +An [earlier attempt with the same pin](../benchmarks/2026-09-29-cellule-main-attempt/README.md) +stopped after ten BeyondDB cases when eight-client TransactGetItems returned +a throttling cancellation. SQLite completed all 24 cases while host load rose +to 90.4 on 12 logical CPUs. A signed 1 KiB PutItem survived an unclean owner +restart, while the larger SDK restart suite separately failed with HTTP 503 +during GSI setup under similar contention. That verification failure remains +open pending a successful repeat. + +## Refresh of the earlier high-throughput sample + +The `1,488.6/1,903.2` GetItem and `81.2/176.2` PutItem requests/s figures previously quoted in PR #14 were reported with commit `880b4aa` from a then-fresh local fixture. We could not locate its raw benchmark JSON. They are historical observations, not verified current-release rates. + +On September 29, 2026, two fresh four-partition RustFS fixtures ran the `af8fab7` release binary with `auth_cache_enabled: true`, 64 seeded items carrying 1 KiB payloads, signed boto3, five-second cases, and SDK retries disabled. In the complete run, GetItem measured **164.13/549.53 requests/s** at one/eight clients, PutItem **10.13/31.56**, and TransactWriteItems **0.74/2.10**. Its 24 cases had no SDK request errors. The other run stopped at eight-client TransactGetItems after eight read timeouts. Both servers logged deferred background work from Cell mailbox-byte exhaustion. Other virtual machines and Rust builds drove the 12-logical-CPU host's load average above 30 during the test. + +This refresh does not reproduce the earlier rates and is not a clean matched regression test: the code revision and host load differ. The [raw results and fixture details](../benchmarks/2026-09-29-claim-refresh/README.md) provide the full API table, latencies, errors, and conditions. Repeat on an otherwise idle host with retained raw results and background-work checks before using either sample as a performance target. + +A later release check with an opt-in persistent follower store also left the +serving path on RustFS publication. It measured GetItem at **272.61/768.80** +requests/s and PutItem at **10.91/45.55** at one/eight clients. All 24 cases +had zero SDK request errors, but the server logged six deferred background +operations and the host was heavily loaded. The [complete API table and raw +results](../benchmarks/2026-09-29-follower-receiver-check/README.md) show that +the historical peaks still are not reproducible on this fixture. Enabling the +receiver alone does not change write durability or imply a throughput gain. + +## Contemporaneous comparison with pinned ExtendDB SQLite + +A later pair of fresh release fixtures exercised all 12 benchmark APIs with the same signed boto3 workload. The [full comparison and raw results](../benchmarks/2026-09-29-sqlite-comparison/README.md) show BeyondDB near SQLite on warm `GetItem` (583 versus 646 requests/s at one client), but far behind on `PutItem` (9 versus 811) and cross-partition `TransactGetItems` (1.5 versus 572). Eight-client `Scan` and one-client `BatchGetItem` were the only cases in which BeyondDB exceeded SQLite. Both backends had zero SDK request errors; BeyondDB also logged deferred background work. + +These sequential cases ran while other virtual machines consumed CPU, and host load changed between fixtures. ExtendDB's SQLite development mode uses open authorization and local-file durability, while BeyondDB verified IAM and waited for RustFS publication. The result identifies the remaining work; it does not establish a controlled throughput ratio or a production capacity target. + +A separate [direct RustFS PUT probe](../benchmarks/2026-09-29-rustfs-put-probe/README.md) +measured object-store calls without the DynamoDB or Cell request path. It ran +under even higher host load, so its rates are diagnostic. It reinforces the +need to measure publication I/O and evaluate Cellule's follower durability +path before expecting local SQLite write latency from this RustFS fixture. The +[follower durability design](follower-durability.md) lists the BeyondDB +integration and recovery gates required before that mode can be benchmarked. ## What a request waits for @@ -18,7 +68,11 @@ signed AWS SDK request With the default `auth_cache_enabled: false`, the server has no process-wide credential or authorization-result cache. It keeps a positive in-memory proof that a credential Cell exists, while it still reads the credential record for every signed request. Table metadata and authorization use pass-through stores so concurrent deletion, recreation, and policy changes are observed. Item routing reads the published directory; writes also wait for durable publication. These choices protect correctness but add work compared with an embedded SQLite test server. The route anchor and leaf can be read concurrently because the anchor remains the authority for whether a route is published. -For a workload that accepts bounded cross-node visibility, set `auth_cache_enabled` to `true` in the node configuration. This enables ExtendDB's 60-second stale-while-revalidate credential, IAM, and table metadata caches and wires local management invalidation through `AuthCacheRegistry`. It can remove several catalog reads from a warm request, but it does not remove the data Cell lookup or durable write publication. Keep it disabled when immediate remote credential revocation or table recreation visibility is required. +For a workload that accepts bounded cross-node visibility, set `auth_cache_enabled` to `true` in the node configuration. This enables ExtendDB's 60-second stale-while-revalidate credential, IAM, and table metadata caches and wires local management invalidation through `AuthCacheRegistry`. It also caches immutable catalog proofs and resident local Cell handles for 500 ms, so warm requests avoid repeated catalog and owner-resolution reads. The Cell handle still fences drained owners; authority is refreshed after the cache window. Keep it disabled when immediate remote credential revocation, table recreation visibility, or owner changes are required. + +The same opt-in mode caches positive `DescribeTable` and `ListTables` responses for 500 ms, with at most 128 entries of each type per node. Local table creation, deletion, and update invalidate these responses as soon as the durable command completes. A remote node's table change can remain absent from a cached response until its entry expires. Item reads and writes still reach their owning Cell. + +On a fresh release fixture, the metadata cache measured 489/571 `DescribeTable` and 583/786 `ListTables` requests/s at one/eight clients, compared with 285/384 and 215/472 in an earlier BeyondDB fixture. A nearby SQLite fixture reached 835/1,196 and 701/343 respectively; its eight-client listing rate fell under host contention. The [raw metadata sample](../benchmarks/2026-09-29-metadata-cache/README.md) records the conditions. These short runs show an improvement in the cached path, not consistent SQLite parity. ## Release comparison: one local node @@ -39,6 +93,52 @@ An opt-in cache sample (`auth_cache_enabled: true`) on a fresh 16 GiB RustFS fix The transaction path overlaps coordinator publication, route discovery, payload reads, participant prepares, and prepare evidence with bounded concurrency. A fresh five-second release sample from the current transaction path completed `TransactGetItems` at 3.04 requests/s with one client (p95 422 ms) and 7.08 requests/s with eight clients (p95 1.41 s). `TransactWriteItems` reached 1.73 requests/s with one client (p95 1.34 s) and 1.51 requests/s with eight clients (p95 6.38 s). The eight-client cases contend on a single account coordinator and use the same closed-loop host, so they show contention behavior rather than a capacity target. All transaction requests succeeded. Transaction throughput remains far below the SQLite baseline and needs coordinator and durable-publication work before it is suitable for large-scale application workloads. +The server now sizes Cellule's SQL worker pool from host parallelism (capped by Cellule at sixteen workers) instead of pinning every node to four workers. On a separate clean RustFS fixture, this configuration measured 117.3 `GetItem` requests/s at one client (p95 10.3 ms) and 559.0 requests/s at eight clients (p95 18.3 ms), 47.6 `UpdateItem` requests/s at one client (p95 35.0 ms) and 121.2 requests/s at eight clients (p95 143.7 ms), and 19.9 `BatchWriteItem` requests/s at one client (p95 102.2 ms) and 59.7 requests/s at eight clients (p95 218.7 ms). `TransactWriteItems` measured 1.56 requests/s at one client and 1.74 requests/s at eight clients; a concurrent eight-client `TransactGetItems` case hit one `ThrottlingError`. These results show better parallel point and batch work on this host, while the transaction bottleneck remains. + +When `auth_cache_enabled` is enabled, routed table directory leaf pages are cached by account and table generation. Large directories can retain up to 64 partial pages instead of caching only a complete page. The owning data Cell still checks the cached epoch, and stale or split routes invalidate the entry; this keeps route changes safe while avoiding a directory traversal on steady-state point operations. The local handle cache is 500 ms. An earlier release fixture measured `GetItem` at 1,323.6 requests/s with one client (p95 1.4 ms) and 1,776.6 requests/s with eight clients (p95 7.4 ms), `Query` at 1,293.9/1,722.2 requests/s (p95 1.4/7.6 ms), `PutItem` at 62.6/120.8 requests/s (p95 25.4/155.2 ms), and `UpdateItem` at 54.4/103.6 requests/s (p95 68.0/170.1 ms). All cases had zero errors. In that earlier run, reads exceeded the one-client SQLite samples; durable writes remained slower because each mutation waited for publication. + +Increasing the local handle cache from 50 ms to 500 ms reduces repeated authority resolution inside a transaction. On a fresh four-partition fixture, signed `TransactWriteItems` reached 4.46 requests/s at one client (p95 306 ms) and 1.55 requests/s at four clients (p95 2.87 s), with zero errors. An eight-client run reached 2.02 requests/s before one `ServiceUnavailable`; the coordinator and durable participant Cells still contend under concurrency. The longer cache remains safe for owner fencing because each resident `CellHandle` rejects drained ownership; it bounds fresh authority discovery at 500 ms when the cache is enabled. + +The same earlier run measured `Scan` at 1,101.1/1,660.0 requests/s, `BatchGetItem` at 857.2/1,498.2 requests/s (1,714.3/2,996.4 items/s), `BatchWriteItem` at 33.8/64.3 requests/s (67.5/128.5 items/s), `DescribeTable` at 919.0/1,474.0 requests/s, and `ListTables` at 1,457.8/1,725.0 requests/s for one/eight clients. Transactions reached 4.17/4.95 `TransactGetItems` requests/s and 1.27/1.53 `TransactWriteItems` requests/s; all cases completed without request errors, but transaction latency and durable-write throughput remain the limiting gap. + +The no-return mutation path now avoids fetching and encoding an old image for an unconditional `PutItem` or `DeleteItem`, and avoids cloning and returning the old image when `UpdateItem` does not request return values. Stream-enabled tables and conditional requests still read the previous image when the protocol requires it. The `880b4aa` report described a fresh release fixture with the same 12-logical-CPU host, four initial partitions, local RustFS, 1 KiB items, signed boto3 requests, and zero errors. In that report, `PutItem` measured 81.2/176.2 requests/s at one/eight clients (p95 15.0/102.3 ms), up from 62.6/120.8 in the preceding handle-cache sample. `UpdateItem` measured 65.4/132.4 requests/s (p95 57.5/134.6 ms), up from 54.4/103.6. The broader sample measured `Scan` at 1,159.5/1,806.9 requests/s, `BatchGetItem` at 898.2/1,602.7 requests/s, `BatchWriteItem` at 32.9/72.3 requests/s, and metadata at 1,028.4โ€“1,957.0 requests/s. `TransactGetItems` reached 4.78/10.92 requests/s and `TransactWriteItems` 2.45/1.55 requests/s; all requests succeeded, but transaction latency remained 209โ€“3,652 ms. That report attributed the write gain to the no-return path. The figures are local comparison measurements rather than capacity claims, and the refreshed run above did not reproduce them. + +The transaction completion path now reuses the coordinator status it already read instead of issuing a second identical status query before participant resolution. On the same fresh release fixture, `TransactWriteItems` measured 1.19 requests/s with one client (p95 1.28 s) and 1.39 requests/s with eight clients (p95 6.69 s), with zero errors. `TransactGetItems` measured 4.55/4.36 requests/s (p95 276/4,048 ms) for one/eight clients. The write improvement is a reduction in coordination overhead, not a change to the durable two-phase protocol; transactions remain far below the file-backed SQLite baseline. + +Unconditional no-return mutations now use a 1 MiB input and 64 KiB result envelope, and successful no-return updates return a completion marker instead of serializing the updated item. Conditional mutations keep the full envelope so a condition failure can still carry the previous image. This reduces the per-request Cell mailbox reservation from roughly 8 MiB to roughly 1 MiB for the common write path. A fresh signed eight-client release smoke run completed 2,679 `PutItem`, 1,949 `UpdateItem`, and 1,044 `BatchWriteItem` requests in 20 seconds with zero request errors; the run recorded 133.64, 97.06, and 51.87 requests/s respectively. These are closed-loop observations on one local fixture, not a production capacity claim. + +Small item images up to the 256 KiB storage chunk now use one bounded SQLite blob update instead of allocating a zeroblob, looking up its rowid, and issuing a chunk write. The isolated signed 1 KiB fixture measured `PutItem` at 52.16/112.86 requests/s and `UpdateItem` at 53.66/93.95 requests/s for one/eight clients, with zero errors. This removes local SQL work; the remaining write latency is still dominated by durable Cell publication to the object store. + +The write path also stores serialized images up to 256 KiB directly in the primary item row. Larger images keep the chunked storage path. This avoids the follow-up blob update for the common small-item case while preserving the same read and recovery format; it does not change the durability boundary or the performance caveat above. + +The high-concurrency fixture also logged deferred background maintenance and transaction-recovery work when the runtime resource ledger filled. That does not invalidate the completed API calls, but it is additional evidence that these numbers are a comparison sample rather than a sustainable capacity target. + +Small transaction payloads now travel inline in the durable phase command, avoiding a separate upload command; payloads near the transfer limit still use the multipart path. A clean fixture measured `TransactGetItems` at 3.37 requests/s with one client (p95 362 ms). The corresponding eight-client run encountered service-unavailable errors under contention, and `TransactWriteItems` measured 1.29 requests/s with one client and 1.43 requests/s with eight clients. This removes avoidable round trips but does not yet close the transaction gap; treat the high-concurrency cases as failure evidence, not capacity results. + +Same-Cell transactions now use one atomic Cell command after routing. This removes the coordinator and participant prepare/resolve round trips when every operation belongs to one account or one installed data Cell; requests that span Cells, or that carry an idempotency token, keep the durable coordinator protocol. On a fresh one-partition fixture with two 1 KiB items per request, signed boto3, local RustFS, and zero request errors, `TransactGetItems` measured 55.28 requests/s at one client (p95 77.49 ms) and 63.08 requests/s at eight clients (p95 202.39 ms). `TransactWriteItems` measured 6.04 requests/s at one client (p95 214.67 ms) and 1.45 requests/s at eight clients (p95 6.30 s). The eight-client write result is contention on one durable Cell, while the one-client result shows the remaining Cell commit cost; these numbers are a targeted fast-path sample, not a fleet capacity target. A four-partition random-key workload still sends most two-item requests through the coordinator. + +Same-Cell `TransactGetItems` now uses a Cellule read-only query instead of a mutation command. The query runs the lock checks and item reads on the Cell worker without writing `sys_requests` or publishing an LTX for a read-only batch; the worker serialization preserves the single-Cell snapshot boundary. On a fresh one-partition release fixture with two 256-byte items per request, signed boto3, local RustFS, and zero errors, this measured 577.49 requests/s at one client (p95 2.86 ms) and 1,134.32 requests/s at eight clients (p95 11.82 ms), or 1,154.97 and 2,268.64 items/s. The preceding same-Cell command path measured 55.28 and 63.08 requests/s, so this removes the read publication cost. The optimization applies only when routing proves one participant; cross-Cell reads still use the durable coordinator protocol, and writes still require durable publication. + +Cross-Cell read cleanup now records all participant release receipts in one coordinator command after the participant Cells have durably released their images. On a fresh four-partition fixture, random two-item signed SDK `TransactGetItems` measured 6.09 requests/s at one client (p95 268 ms) and 4.52 requests/s at eight clients (p95 6.46 s), with zero errors. The preceding equivalent run measured 5.58/3.43 requests/s and returned one eight-client `ServiceUnavailable`; this is a targeted stability and round-trip reduction, not evidence of SQLite-parity transaction capacity. + +Cross-Cell prepare and terminal-resolution receipts now use bounded coordinator batch commands after participant Cell work completes concurrently. A fresh four-partition fixture measured `TransactGetItems` at 5.71/4.28 requests/s for one/eight clients and `TransactWriteItems` at 1.17/1.26 requests/s, with zero errors. This removes one coordinator command per participant while retaining replay checks; the new fixture does not show a throughput increase, and durable participant publication remains the transaction bottleneck. + +The route-cache lock path is now synchronous: the process-local route and account-placement maps use one short `std::sync::RwLock` critical section instead of awaiting Tokio locks for every routed request. A matched five-second comparison used the previous 500 ms-handle-cache release and the current release, fresh four-partition tables, local RustFS, signed boto3 requests, 256-byte items, 64 seeded keys, and one/eight clients. Every case completed without errors. The current release improved eight-client `Scan` from 1,249 to 1,457 requests/s (+16.6%), `Query` from 1,430 to 1,509 (+5.5%), and `BatchGetItem` from 1,195 to 1,244 (+4.1%); one-client `Query` improved 11.1%. `PutItem` improved 12.4โ€“13.4% (51.7โ†’58.2 and 95.3โ†’108.1 requests/s), while `BatchWriteItem` and `TransactWriteItems` changed by less than measurement noise (โˆ’2.2% to +3.5% and โˆ’5.4% to +2.9%). Durable publication remains the dominant write and transaction cost, so this optimization does not establish SQLite parity or a production capacity target. + +The node now exposes `max_active_cells` instead of fixing the runtime admission ceiling at 64. A clean 64-partition table could be provisioned with `max_active_cells: 128`, whereas the previous ceiling left the table in `CREATING` after capacity exhaustion. On the 64-partition fixture, random-key `PutItem` reached 20.4/48.3/58.8 requests/s at 1/8/32 clients and fell to 46.8 at 64 clients; `UpdateItem` reached 14.3/14.1/54.8 requests/s at 1/8/32 clients, and the 64-client run returned five `ServiceUnavailable` errors. More resident Cells remove the admission ceiling but do not remove the 16-worker and durable-publication bottlenecks, so this is a capacity-control improvement rather than evidence of SQLite-parity throughput. + +The serving binary now accepts an optional `sql_workers` override. Without it, Cellule derives the worker count from host parallelism and caps it at sixteen; setting it explicitly is useful on a multi-partition node when the host has spare CPU and memory. Workers own their SQLite connections and improve scheduling between independent Cells, but each Cell still executes commands in order and publishes one durable root at a time. This knob therefore addresses partition fan-out and worker undersizing, not the per-Cell object-publication ceiling. Record the chosen value with benchmark results and verify resident memory before increasing it. + +A matched exploratory comparison on fresh four-partition RustFS fixtures (256-byte items, signed boto3, five-second cases) did not show a consistent write gain from forcing sixteen workers over the host-derived twelve: `PutItem` was 29.4/79.5 versus 38.2/62.1 requests/s at 1/8 clients, `UpdateItem` was 35.4/41.0 versus 28.8/75.7, and `BatchWriteItem` was 9.9/15.9 versus 13.7/26.0. The fixtures are too short to establish a capacity target; retain the override for measured multi-Cell workloads, not as a general single-Cell write optimization. + +A fresh release sample on 2026-09-29 measured the current no-return delete path on a four-partition RustFS fixture with `max_active_cells: 128`, `sql_workers: 12`, `auth_cache_enabled: true`, signed boto3 requests, 256-byte items, and five-second cases. All requests completed without errors. `PutItem` reached 15.0/22.7 requests/s at 1/8 clients (p95 94.9/611.0 ms), `UpdateItem` 15.5/20.6 (p95 106.4/780.3 ms), and unconditional absent-key `DeleteItem` 15.9/21.7 (p95 93.5/776.7 ms). `BatchWriteItem` reached 7.4/10.6 requests/s, or 14.9/21.2 items/s, at 1/8 clients (p95 192.8/1,152.5 ms). The delete case used a unique key that was absent, so it exercises the no-return envelope without an old image. This run confirms zero-error behavior for the release binary and the reduced delete work, while durable publication remains the dominant cost and these write rates remain below the file-backed SQLite baseline. + +A 64-partition release fixture after this change reached 11.9/19.6/23.4/21.1 `PutItem` requests/s at 1/8/32/64 clients and 11.4/21.3/19.2/19.2 `UpdateItem` requests/s, all with zero errors. The result did not materially improve on the previous 64-partition run, which confirms that directory traversal is not the dominant write cost at this scale; Cell command publication and its object-store round trips remain the next optimization boundary. + +The routed no-return path now coalesces a bounded burst of unconditional `PutItem`, `UpdateItem`, and `DeleteItem` requests for the same partition into one partition-local durable command. The batcher waits up to 2 ms for a partial queue, flushes a full queue immediately, accepts at most 16 distinct keys, and keeps conditional, old-image, account-local, and coordinator transaction requests on their existing paths. On fresh four-partition RustFS fixtures with 256-byte items, signed boto3, `max_active_cells: 128`, `sql_workers: 12`, and zero errors, the short baseline/current comparisons at one/eight clients were: `PutItem` 14.1/22.4 versus 12.9/36.4 requests/s, missing-key `DeleteItem` 13.6/20.3 versus 14.3/32.5, `UpdateItem` 13.7/18.6 versus 14.1/20.3, and two-item `BatchWriteItem` 13.8/19.7 versus 13.0/35.0 items/s. The short single-client samples vary with object-store timing; the larger concurrent gains come from sharing one publication across requests. Update expressions still spend more time in SQLite/index work, so their gain is smaller. This is a local-node optimization, not a fleet capacity claim. + +The pinned ExtendDB `BatchWriteItem` handler awaits each item mutation in a request before starting the next. This means a single two-item request still incurs two durable publications when both items are handled individually. The no-return batcher can combine mutations arriving from *different concurrent requests*, but it cannot combine items that ExtendDB submits sequentially within one request. This is a separate bottleneck from the batcher's queue window; improving it requires a batch-aware protocol path that preserves ExtendDB's validation and result semantics. + Temporary stage timings on a separate instrumented fixture put typical credential, IAM, table record, and data Cell reads around 3โ€“4 ms each, while each route traversal took around 6โ€“7 ms. The requests perform several of these operations in sequence. The instrumented write fixture differed materially from the clean fixture, so its write timings are not a publication-cost estimate. A temporary S3 proxy disrupted publication and its counts were discarded. ## Run a repeatable point-operation sample diff --git a/docs/user-guide.md b/docs/user-guide.md index f7d5d19..5235cd9 100644 --- a/docs/user-guide.md +++ b/docs/user-guide.md @@ -47,7 +47,7 @@ aws dynamodb describe-table \ --endpoint-url "$BEYONDDB_ENDPOINT" ``` -The serving binary provisions `initial_partitions` data Cells for a new routed table. `DescribeTable` can report `CREATING` until range publication finishes. Changing `initial_partitions` later affects new table generations only. +The serving binary provisions `initial_partitions` data Cells for a new routed table. `DescribeTable` can report `CREATING` until range publication finishes. Changing `initial_partitions` later affects new table generations only. The node's `max_active_cells` budget must include those data Cells plus account, coordinator, and management Cells; if the budget is too small, provisioning remains pending until capacity is available. `sql_workers` is an optional override for the SQL worker count (maximum sixteen); the default follows host parallelism. Use it for multi-partition workloads after measuring CPU and memory headroom. It improves independent Cell scheduling, while a single hot Cell remains serialized for ordering and durable publication. ## Write and read an item diff --git a/scripts/bench.py b/scripts/bench.py index e6020c1..1b548a8 100644 --- a/scripts/bench.py +++ b/scripts/bench.py @@ -218,6 +218,11 @@ def main(): parser.add_argument("--sort-key-value", default="1") parser.add_argument("--operations", nargs="+", choices=OPERATIONS, default=["get", "put"]) parser.add_argument("--output", help="write JSON results to this path") + parser.add_argument( + "--continue-on-error", + action="store_true", + help="record every case even if an earlier case has request errors", + ) args = parser.parse_args() if args.seconds <= 0 or args.payload_bytes <= 0 or args.seed_keys <= 0: parser.error("seconds, payload bytes, and seed keys must be positive") @@ -287,8 +292,10 @@ def main(): with open(args.output, "w", encoding="utf-8") as output: json.dump(results, output, indent=2) output.write("\n") - if case["errors"]: + if case["errors"] and not args.continue_on_error: raise SystemExit("stopped after request error") + if any(case["errors"] for case in results["cases"]): + raise SystemExit("one or more cases had request errors") if __name__ == "__main__": diff --git a/src/backend.rs b/src/backend.rs index 5df8960..533dd46 100644 --- a/src/backend.rs +++ b/src/backend.rs @@ -1,8 +1,10 @@ //! ExtendDB table operations routed to account Cells. mod admission; +mod batch; mod data; mod global_index; +mod metadata_cache; mod recovery; mod remaining; mod statistics; @@ -14,11 +16,17 @@ mod transaction_read; mod transaction_transport; use std::{ - collections::HashSet, - sync::Arc, + collections::{HashMap, HashSet}, + sync::{Arc, RwLock}, time::{SystemTime, UNIX_EPOCH}, }; +use super::{ + APPLICATION, CreateTable, CreateTableOutcome, DeleteTable, DeleteTableOutcome, DescribeTable, + DescribeTableById, Json, ListTables, ListTablesInput, ListTablesOutcome, NAMESPACE, + PartitionSpec, RoutePageInput, RoutePageOutcome, TablePlacement, TableRecord, TableSpec, + TableUpdate, UpdateTable, UpdateTableOutcome, account_target, +}; use cellule_runtime::client::{CellClient, InvocationError, ReadPolicy}; use cellule_runtime::identity::{CellTarget, RequestId, TenantId}; use cellule_runtime::{MutationIdentity, partition_for_shard}; @@ -31,14 +39,9 @@ use extenddb_core::types::{ }; use extenddb_storage::error::StorageError; use extenddb_storage::{BoxedFuture, TableEngine}; -use tokio::sync::RwLock; -use super::{ - APPLICATION, CreateTable, CreateTableOutcome, DeleteTable, DeleteTableOutcome, DescribeTable, - DescribeTableById, Json, ListTables, ListTablesInput, ListTablesOutcome, NAMESPACE, - PartitionSpec, RoutePageInput, RoutePageOutcome, TablePlacement, TableRecord, TableSpec, - TableUpdate, UpdateTable, UpdateTableOutcome, account_target, -}; +use batch::NoReturnBatcher; +use metadata_cache::MetadataCache; /// Installs an initial table's data Cells before its route becomes visible. pub trait InitialPartitionProvisioner: Send + Sync { @@ -86,12 +89,29 @@ pub trait CoordinatorProvisioner: Send + Sync { /// ExtendDB table backend over already-provisioned and routable account Cells. pub struct CellStorage { client: CellClient, + no_return_batcher: Arc, region: String, initial_partitions: Option>, coordinators: Option>, - account_placement_cache: Arc>>, + route_cache: Arc>, + route_cache_enabled: bool, + metadata_cache: MetadataCache, +} + +#[derive(Clone)] +struct CachedRoute { + partitions: Vec, +} + +struct RouteCacheState { + account_placement: HashSet, + routes: HashMap<(String, String), Vec>, } +// Keep a bounded set of directory leaf pages for large routed tables. A stale +// epoch still fences the data Cell and invalidates the complete table entry. +const MAX_CACHED_ROUTE_PAGES: usize = 64; + impl CellStorage { pub(crate) fn client(&self) -> &CellClient { &self.client @@ -103,12 +123,19 @@ impl CellStorage { /// its requests through this backend. All reads use the current owner so /// transaction decisions and prepared intents cannot come from stale snapshots. pub fn new(client: CellClient, region: impl Into) -> Self { + let client = client.with_read_policy(ReadPolicy::CurrentOwner); Self { - client: client.with_read_policy(ReadPolicy::CurrentOwner), + no_return_batcher: Arc::new(NoReturnBatcher::new(client.clone())), + client, region: region.into(), initial_partitions: None, coordinators: None, - account_placement_cache: Arc::new(RwLock::new(HashSet::new())), + route_cache: Arc::new(RwLock::new(RouteCacheState { + account_placement: HashSet::new(), + routes: HashMap::new(), + })), + route_cache_enabled: false, + metadata_cache: MetadataCache::default(), } } @@ -131,6 +158,119 @@ impl CellStorage { self.initial_partitions = Some(provisioner); self } + + /// Enables process-local route pages and short-lived metadata responses. + /// + /// A stale cached route is rejected by the data Cell and invalidated; the + /// next request reads the current directory. Metadata responses expire + /// after 500 ms and local table changes invalidate them. Keep this opt-in + /// alongside other explicitly stale-tolerant serving caches. + #[must_use] + pub fn with_route_cache(mut self, enabled: bool) -> Self { + self.route_cache_enabled = enabled; + self + } + + pub(super) fn invalidate_route_cache(&self, account_id: &str, table_id: &str) { + let key = (account_id.to_owned(), table_id.to_owned()); + match self.route_cache.write() { + Ok(mut cache) => { + cache.routes.remove(&key); + cache.account_placement.remove(table_id); + } + Err(poisoned) => { + let mut cache = poisoned.into_inner(); + cache.routes.remove(&key); + cache.account_placement.remove(table_id); + } + } + } + + pub(super) fn account_placement_cached(&self, table_id: &str) -> bool { + match self.route_cache.read() { + Ok(cache) => cache.account_placement.contains(table_id), + Err(poisoned) => poisoned.into_inner().account_placement.contains(table_id), + } + } + + pub(super) fn cached_route( + &self, + key: &(String, String), + hash: [u8; 16], + ) -> Option<([u8; 16], u64)> { + let lookup = |cache: &RouteCacheState| { + cache.routes.get(key).and_then(|pages| { + pages.iter().find_map(|page| { + page.partitions + .iter() + .find(|partition| { + hash >= partition.lower + && partition.upper.is_none_or(|upper| hash < upper) + }) + .map(|partition| (partition.partition_id, partition.epoch)) + }) + }) + }; + match self.route_cache.read() { + Ok(cache) => lookup(&cache), + Err(poisoned) => lookup(&poisoned.into_inner()), + } + } + + pub(super) fn cache_account_placement(&self, table_id: String) { + match self.route_cache.write() { + Ok(mut cache) => { + cache.account_placement.insert(table_id); + } + Err(poisoned) => { + poisoned.into_inner().account_placement.insert(table_id); + } + } + } + + pub(super) fn cache_route( + &self, + key: (String, String), + partitions: Vec, + complete: bool, + ) { + match self.route_cache.write() { + Ok(mut cache) => { + cache_route_pages(&mut cache, key, partitions, complete); + } + Err(poisoned) => { + cache_route_pages(&mut poisoned.into_inner(), key, partitions, complete); + } + } + } +} + +fn cache_route_pages( + cache: &mut RouteCacheState, + key: (String, String), + partitions: Vec, + complete: bool, +) { + let pages = cache.routes.entry(key).or_default(); + if complete { + pages.clear(); + pages.push(CachedRoute { partitions }); + return; + } + let Some(first_lower) = partitions.first().map(|partition| partition.lower) else { + return; + }; + if let Some(page) = pages + .iter_mut() + .find(|page| page.partitions.first().map(|partition| partition.lower) == Some(first_lower)) + { + page.partitions = partitions; + return; + } + if pages.len() >= MAX_CACHED_ROUTE_PAGES { + pages.remove(0); + } + pages.push(CachedRoute { partitions }); } impl TableEngine for CellStorage { @@ -256,6 +396,7 @@ impl TableEngine for CellStorage { }, Err(error) => return Err(cell_error(error)), }; + self.metadata_cache.invalidate(); if let Some(provisioner) = &self.initial_partitions { table_creation::publish_initial_routes( provisioner.as_ref(), @@ -330,10 +471,8 @@ impl TableEngine for CellStorage { }, Err(error) => return Err(cell_error(error)), }; - self.account_placement_cache - .write() - .await - .remove(&previous.id); + self.metadata_cache.invalidate(); + self.invalidate_route_cache(&account_id, &previous.id); // A concurrent delete/recreate can change the name's generation. // Never attach the previous table's sample to the newly deleted one. let same_generation = record.id == previous.id; @@ -352,6 +491,15 @@ impl TableEngine for CellStorage { ) -> BoxedFuture<'_, Result> { let account_id = account_id.to_owned(); Box::pin(async move { + let name = input.table_name.clone(); + if self.route_cache_enabled + && let Some(cached) = self.metadata_cache.description(&account_id, &name) + { + return Ok(cached); + } + let generation = self + .route_cache_enabled + .then(|| self.metadata_cache.generation()); let (record, status) = match self.lifecycle(&account_id, &input.table_name).await? { crate::TableLifecycle::Missing => { return Err(StorageError::TableNotFound(input.table_name)); @@ -368,7 +516,16 @@ impl TableEngine for CellStorage { (record, status) } }; - self.table_description(record, &account_id, status).await + let description = self.table_description(record, &account_id, status).await?; + if let Some(generation) = generation { + self.metadata_cache.insert_description( + &account_id, + name, + generation, + description.clone(), + ); + } + Ok(description) }) } @@ -379,6 +536,16 @@ impl TableEngine for CellStorage { ) -> BoxedFuture<'_, Result> { let account_id = account_id.to_owned(); Box::pin(async move { + let limit = i64::from(input.limit.unwrap_or(100)); + let start = input.exclusive_start_table_name; + if self.route_cache_enabled + && let Some(cached) = self.metadata_cache.listing(&account_id, limit, &start) + { + return Ok(cached); + } + let generation = self + .route_cache_enabled + .then(|| self.metadata_cache.generation()); let target = target(&account_id)?; let output = self .client @@ -386,18 +553,30 @@ impl TableEngine for CellStorage { &target, None, Json(ListTablesInput { - limit: i64::from(input.limit.unwrap_or(100)), - exclusive_start: input.exclusive_start_table_name, + limit, + exclusive_start: start.clone(), live_only: false, }), ) .await .map_err(cell_error)?; match output.output.0 { - ListTablesOutcome::Page(page) => Ok(ListTablesOutput { - table_names: page.names, - last_evaluated_table_name: page.last_evaluated, - }), + ListTablesOutcome::Page(page) => { + let listing = ListTablesOutput { + table_names: page.names, + last_evaluated_table_name: page.last_evaluated, + }; + if let Some(generation) = generation { + self.metadata_cache.insert_listing( + &account_id, + limit, + start, + generation, + listing.clone(), + ); + } + Ok(listing) + } ListTablesOutcome::InvalidLimit => Err(StorageError::Validation( "table listing limit must be 1..=100".into(), )), @@ -476,6 +655,7 @@ impl TableEngine for CellStorage { }, Err(error) => return Err(cell_error(error)), }; + self.metadata_cache.invalidate(); self.table_description(record, &account_id, TableStatus::Active) .await }) diff --git a/src/backend/admission.rs b/src/backend/admission.rs index 18aeebd..262ff50 100644 --- a/src/backend/admission.rs +++ b/src/backend/admission.rs @@ -13,13 +13,14 @@ use crate::{ BeginCrossCellTransaction, BeginCrossCellTransactionInput, BeginCrossCellTransactionOutcome, CoordinatorDecision, CoordinatorParticipant, CoordinatorParticipantTarget, IndexedTransactionOperation, Json, ReadCoordinatorToken, ReadCoordinatorTokenOutcome, - ReadCrossCellTransaction, ReadCrossCellTransactionInput, TransactionOperation, - TransactionToken, coordinator_target, + ReadCrossCellTransaction, ReadCrossCellTransactionInput, TransactionCommandInput, + TransactionOperation, TransactionToken, coordinator_target, }; pub(super) struct AdmittedTransaction { pub identity: ReadCrossCellTransactionInput, pub decision: CoordinatorDecision, + pub participant_count: u8, pub replay: bool, } @@ -87,7 +88,7 @@ impl CellStorage { .await } - async fn route_transaction_participants( + pub(super) async fn route_transaction_participants( &self, account_id: &str, operations: Vec, @@ -155,13 +156,30 @@ impl CellStorage { participants: participants.into_values().collect(), }; let identity = mutation_identity()?; - let reference = self - .upload_transaction::(&coordinator, identity, &input) - .await?; - let result = self - .client - .command::(&coordinator, identity, Json(reference)) - .await; + let inline = serde_json::to_vec(&input) + .map_err(|error| StorageError::Internal(error.to_string()))? + .len() + <= crate::transaction_transport::INLINE_BYTES; + let result = if inline { + self.client + .command::( + &coordinator, + identity, + Json(TransactionCommandInput::Inline(input)), + ) + .await + } else { + let reference = self + .upload_transaction::(&coordinator, identity, &input) + .await?; + self.client + .command::( + &coordinator, + identity, + Json(TransactionCommandInput::Reference(reference)), + ) + .await + }; let (transaction_id, prior) = match result { Ok(result) => match result.output.0 { BeginCrossCellTransactionOutcome::Begun => { @@ -225,8 +243,8 @@ impl CellStorage { transaction_id: [u8; 16], prior: CoordinatorDecision, ) -> Result { - let decision = self - .resume_cross_cell_transaction(account_id, &routing_key, transaction_id) + let status = self + .resume_cross_cell_transaction_status(account_id, &routing_key, transaction_id) .await?; Ok(AdmittedTransaction { identity: ReadCrossCellTransactionInput { @@ -234,7 +252,8 @@ impl CellStorage { routing_key, transaction_id, }, - decision, + decision: status.decision, + participant_count: status.participant_count, replay: prior == CoordinatorDecision::Commit, }) } diff --git a/src/backend/batch.rs b/src/backend/batch.rs new file mode 100644 index 0000000..e7c6913 --- /dev/null +++ b/src/backend/batch.rs @@ -0,0 +1,225 @@ +//! Bounded coalescing for idempotent routed mutations. + +use std::collections::{HashMap, VecDeque}; +use std::sync::{ + Arc, Weak, + atomic::{AtomicBool, Ordering}, +}; +use std::time::Duration; + +use cellule_runtime::client::InvocationError; +use cellule_runtime::identity::CellTarget; +use extenddb_storage::error::StorageError; +use tokio::sync::{Mutex, oneshot}; + +use super::{cell_error, mutation_identity}; +use crate::{ + Json, PartitionTransactWriteInput, PartitionTransactWriteNoReturn, + PartitionTransactWriteOutcome, TransactionFailure, TransactionOperation, +}; + +const BATCH_WINDOW: Duration = Duration::from_millis(2); +const MAX_BATCH_OPERATIONS: usize = 16; + +/// An unconditional mutation that can be replayed safely as part of a batch. +#[derive(Clone)] +pub(crate) struct NoReturnMutation { + /// Table generation checked by the partition command. + pub table_id: String, + /// Directory epoch checked by the partition command. + pub epoch: u64, + /// Canonical key bytes used to avoid duplicate operations in one batch. + pub key: Vec, + /// The already validated operation to stage in the partition transaction. + pub operation: TransactionOperation, +} + +struct PendingMutation { + mutation: NoReturnMutation, + reply: oneshot::Sender>, +} + +struct Slot { + target: CellTarget, + queued: Mutex>, + scheduled: AtomicBool, +} + +/// Coalesces a small number of idempotent writes before one durable command. +pub(crate) struct NoReturnBatcher { + client: cellule_runtime::CellClient, + slots: Mutex>>, +} + +impl super::CellStorage { + pub(crate) async fn submit_no_return( + &self, + target: CellTarget, + mutation: NoReturnMutation, + ) -> Result<(), StorageError> { + self.no_return_batcher.submit(target, mutation).await + } +} + +impl NoReturnBatcher { + pub(crate) fn new(client: cellule_runtime::CellClient) -> Self { + Self { + client, + slots: Mutex::new(HashMap::new()), + } + } + + pub(crate) async fn submit( + self: &Arc, + target: CellTarget, + mutation: NoReturnMutation, + ) -> Result<(), StorageError> { + let (reply, result) = oneshot::channel(); + let key = *target.cell_id().as_bytes(); + let slot = { + let mut slots = self.slots.lock().await; + if let Some(slot) = slots.get(&key).and_then(Weak::upgrade) { + slot + } else { + let slot = Arc::new(Slot { + target: target.clone(), + queued: Mutex::new(VecDeque::new()), + scheduled: AtomicBool::new(false), + }); + slots.insert(key, Arc::downgrade(&slot)); + slot + } + }; + let should_schedule = { + let mut queued = slot.queued.lock().await; + queued.push_back(PendingMutation { mutation, reply }); + !slot.scheduled.swap(true, Ordering::AcqRel) + }; + if should_schedule { + let batcher = Arc::clone(self); + tokio::spawn(async move { + batcher.flush_slot(slot).await; + }); + } + result + .await + .map_err(|_| StorageError::Transient("batched mutation worker stopped".into()))? + } + + async fn flush_slot(self: Arc, slot: Arc) { + loop { + // A full queue has already had its chance to coalesce. In + // particular, avoid adding a fresh window after the preceding + // durable command completed while more writers were waiting. + if slot.queued.lock().await.len() < MAX_BATCH_OPERATIONS { + tokio::time::sleep(BATCH_WINDOW).await; + } + let batch = { + let mut queued = slot.queued.lock().await; + let Some(first) = queued.front() else { + slot.scheduled.store(false, Ordering::Release); + return; + }; + let table_id = first.mutation.table_id.clone(); + let epoch = first.mutation.epoch; + let mut keys = Vec::new(); + let mut selected = Vec::new(); + let mut deferred = VecDeque::new(); + let queued_len = queued.len(); + while selected.len() < MAX_BATCH_OPERATIONS && !queued.is_empty() { + let Some(candidate) = queued.front() else { + break; + }; + if candidate.mutation.table_id != table_id || candidate.mutation.epoch != epoch + { + break; + } + let Some(candidate) = queued.pop_front() else { + break; + }; + if keys + .iter() + .any(|key: &Vec| key == &candidate.mutation.key) + { + // Keep repeated keys for a later command, but continue + // collecting independent keys behind them. Reordering + // concurrent requests for different items is already + // allowed; requests for one item stay FIFO. + deferred.push_back(candidate); + } else { + keys.push(candidate.mutation.key.clone()); + selected.push(candidate); + } + if selected.len() + deferred.len() >= queued_len { + break; + } + } + while let Some(candidate) = deferred.pop_back() { + queued.push_front(candidate); + } + (table_id, epoch, selected) + }; + let (table_id, epoch, pending) = batch; + if pending.is_empty() { + continue; + } + let (operations, replies): (Vec<_>, Vec<_>) = pending + .into_iter() + .map(|pending| (pending.mutation.operation, pending.reply)) + .unzip(); + let result = self + .client + .command::( + &slot.target, + match mutation_identity() { + Ok(identity) => identity, + Err(error) => { + for reply in replies { + let _ = reply.send(Err(error.clone())); + } + continue; + } + }, + Json(PartitionTransactWriteInput { + table_id, + epoch, + operations, + }), + ) + .await; + let outcome = match result { + Ok(committed) => partition_outcome(committed.output.0), + Err(InvocationError::Rejected(committed)) => partition_outcome(committed.output.0), + Err(error) => Err(cell_error(error)), + }; + for reply in replies { + let _ = reply.send(outcome.clone()); + } + } + } +} + +fn partition_outcome(outcome: PartitionTransactWriteOutcome) -> Result<(), StorageError> { + match outcome { + PartitionTransactWriteOutcome::Applied => Ok(()), + PartitionTransactWriteOutcome::NotInstalled + | PartitionTransactWriteOutcome::StaleRoute + | PartitionTransactWriteOutcome::Sealed + | PartitionTransactWriteOutcome::NotReady + | PartitionTransactWriteOutcome::WrongPartition => Err(StorageError::Transient( + "table partition route changed; retry the request".into(), + )), + PartitionTransactWriteOutcome::Rejected { reason, .. } => match reason { + TransactionFailure::Throttled => Err(StorageError::LimitExceeded( + "partition capacity is exhausted".into(), + )), + TransactionFailure::Conflict => Err(StorageError::TransactionConflict( + "item is locked by a transaction".into(), + )), + TransactionFailure::Validation(message) => Err(StorageError::Validation(message)), + TransactionFailure::ConditionFailed(_) => Err(StorageError::Internal( + "unconditional partition mutation reported a condition failure".into(), + )), + }, + } +} diff --git a/src/backend/data.rs b/src/backend/data.rs index 9a29e24..8ac8b24 100644 --- a/src/backend/data.rs +++ b/src/backend/data.rs @@ -5,6 +5,8 @@ mod routed; +use super::batch::NoReturnMutation; +use super::transaction_read::validate_read_size; use routed::{ partition_delete_rejection, partition_put_rejection, partition_update_rejection, stale_partition, @@ -27,18 +29,120 @@ use super::{CellStorage, cell_error, mutation_identity, target}; use crate::TransactionToken; use crate::expression_wire::{WireCondition, WireUpdate}; use crate::{ - ConditionCheckInput, DeleteItem, DeleteItemInput, GetItem, GetItemInput, GetItemOutcome, - ItemMutationOutcome, Json, PartitionDelete, PartitionDeleteInput, PartitionDeleteOutcome, - PartitionGet, PartitionGetInput, PartitionGetOutcome, PartitionPut, PartitionPutInput, - PartitionPutOutcome, PartitionQuery, PartitionQueryInput, PartitionQueryOutcome, - PartitionUpdate, PartitionUpdateInput, PartitionUpdateOutcome, PutItem, PutItemInput, - ScanItems, ScanItemsInput, ScanItemsOutcome, SortComparison, SortPredicate, TransactionFailure, - TransactionOperation, UpdateItem, UpdateItemInput, UpdateItemOutcome, data_key_hash, + ConditionCheckInput, DeleteItem, DeleteItemInput, DeleteItemNoReturn, GetItem, GetItemInput, + GetItemOutcome, ItemMutationOutcome, Json, PartitionDelete, PartitionDeleteInput, + PartitionDeleteOutcome, PartitionGet, PartitionGetInput, PartitionGetOutcome, PartitionPut, + PartitionPutInput, PartitionPutOutcome, PartitionQuery, PartitionQueryInput, + PartitionQueryOutcome, PartitionTransactReadOutcome, PartitionTransactReadQuery, + PartitionTransactWrite, PartitionTransactWriteInput, PartitionTransactWriteNoReturn, + PartitionTransactWriteOutcome, PartitionUpdate, PartitionUpdateInput, PartitionUpdateOutcome, + PutItem, PutItemInput, PutItemNoReturn, ScanItems, ScanItemsInput, ScanItemsOutcome, + SortComparison, SortPredicate, TransactReadQuery, TransactWrite, TransactWriteInput, + TransactWriteNoReturn, TransactionFailure, TransactionOperation, TransactionOutcome, + TransactionReadOutcome, UpdateItem, UpdateItemInput, UpdateItemNoReturn, UpdateItemOutcome, + data_key_hash, }; use cellule_runtime::client::InvocationError; use cellule_runtime::identity::CellTarget; impl CellStorage { + async fn try_local_transaction_read( + &self, + account_id: &str, + inputs: Vec, + routing: Vec<(TableKeyInfo, Item)>, + ) -> Result>>, StorageError> { + let count = inputs.len(); + let operations = inputs + .into_iter() + .map(TransactionOperation::Read) + .collect::>(); + let participants = self + .route_transaction_participants(account_id, operations.clone(), routing) + .await?; + if participants.len() != 1 { + return Ok(None); + } + let Some((_, participant)) = participants.into_iter().next() else { + return Ok(None); + }; + let mut participant_operations = participant.operations; + participant_operations.sort_unstable_by_key(|operation| operation.index); + let participant_operations = participant_operations + .into_iter() + .map(|operation| match operation.operation { + TransactionOperation::Read(input) => Ok(input), + _ => Err(()), + }) + .collect::, _>>(); + let Ok(participant_operations) = participant_operations else { + return Ok(None); + }; + if participant_operations.len() != count { + return Ok(None); + } + let transaction_operations = participant_operations + .into_iter() + .map(TransactionOperation::Read) + .collect(); + match participant.target { + crate::CoordinatorParticipantTarget::Account => { + let target = target(account_id)?; + let result = self + .client + .query::( + &target, + None, + Json(TransactWriteInput { + operations: transaction_operations, + }), + ) + .await; + match result { + Ok(committed) => match committed.output.0 { + TransactionReadOutcome::Applied(items) => Ok(Some(items)), + TransactionReadOutcome::Rejected { index, reason } => { + Err(transaction_canceled(index, reason, count, &[])) + } + }, + Err(InvocationError::NotStarted(cellule_runtime::Error::Codec( + cellule_runtime::codec::CodecError::Limit, + ))) => Ok(None), + Err(error) => Err(cell_error(error)), + } + } + crate::CoordinatorParticipantTarget::Data { + table_id, + partition_id, + epoch, + } => { + let target = crate::data_target(account_id, &table_id, &partition_id) + .map_err(|error| StorageError::Internal(error.to_string()))?; + let result = self + .client + .query::( + &target, + None, + Json(PartitionTransactWriteInput { + table_id, + epoch, + operations: transaction_operations, + }), + ) + .await; + match result { + Ok(committed) => { + local_partition_transaction_read_result(committed.output.0, count) + } + Err(InvocationError::NotStarted(cellule_runtime::Error::Codec( + cellule_runtime::codec::CodecError::Limit, + ))) => Ok(None), + Err(error) => Err(cell_error(error)), + } + } + } + } + pub(super) async fn delete_expired_partition_item( &self, owner: &CellTarget, @@ -94,55 +198,146 @@ impl DataEngine for CellStorage { Box::pin(async move { let key = extract_key(&item, &key_info.base_key_schema); if let Some((partition, epoch)) = self.routed_owner(&key_info, &key).await? { - let outcome = self - .client - .command::( - &partition, - mutation_identity()?, - Json(PartitionPutInput { - table_id: key_info.table_id, - epoch, - item, - condition, - }), - ) - .await; - let old = match outcome { - Ok(committed) => match committed.output.0 { - PartitionPutOutcome::Applied(old) => old, - _ => return Err(StorageError::Internal("unexpected partition put".into())), - }, - Err(InvocationError::Rejected(committed)) => { - return Err(partition_put_rejection(committed.output.0)); + let no_return = !return_old && condition.is_none(); + if no_return { + let dedup_key = crate::item_key(&key, &key_info.base_key_schema) + .map_err(|error| StorageError::Internal(error.to_string()))?; + let result = self + .submit_no_return( + partition, + NoReturnMutation { + table_id: key_info.table_id.clone(), + epoch, + key: dedup_key, + operation: TransactionOperation::Put(PutItemInput { + table_name: key_info.table_name.clone(), + table_id: key_info.table_id.clone(), + item, + condition: None, + }), + }, + ) + .await; + if result.is_err() { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + } + result?; + return Ok(None); + } + let input = Json(PartitionPutInput { + table_id: key_info.table_id.clone(), + epoch, + item, + condition, + }); + let old = if return_old { + match self + .client + .command::(&partition, mutation_identity()?, input) + .await + { + Ok(committed) => match committed.output.0 { + PartitionPutOutcome::Applied(old) => old, + _ => { + return Err(StorageError::Internal( + "unexpected partition put".into(), + )); + } + }, + Err(InvocationError::Rejected(committed)) => { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + return Err(partition_put_rejection(committed.output.0)); + } + Err(error) => return Err(cell_error(error)), + } + } else { + match self + .client + .command::(&partition, mutation_identity()?, input) + .await + { + Ok(committed) => match committed.output.0 { + PartitionPutOutcome::Applied(_) => None, + _ => { + return Err(StorageError::Internal( + "unexpected partition put result".into(), + )); + } + }, + Err(InvocationError::Rejected(committed)) => { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + return Err(partition_put_rejection(committed.output.0)); + } + Err(error) => return Err(cell_error(error)), } - Err(error) => return Err(cell_error(error)), }; return Ok(if return_old { old } else { None }); } let target = target(&key_info.account_id)?; + let no_return = !return_old && condition.is_none(); let input = PutItemInput { table_name: key_info.table_name.clone(), - table_id: key_info.table_id, + table_id: key_info.table_id.clone(), item, condition, }; - let old = match self - .client - .command::(&target, mutation_identity()?, Json(input)) - .await - { - Ok(committed) => match committed.output.0 { - ItemMutationOutcome::Applied(old) => old, - _ => { - return Err(StorageError::Internal( - "unexpected successful put result".into(), - )); + let old = if return_old { + match self + .client + .command::(&target, mutation_identity()?, Json(input)) + .await + { + Ok(committed) => match committed.output.0 { + ItemMutationOutcome::Applied(old) => old, + _ => { + return Err(StorageError::Internal( + "unexpected successful put result".into(), + )); + } + }, + Err(InvocationError::Rejected(committed)) => { + return Err(mutation_rejection(committed.output.0, &key_info.table_name)); } - }, - Err(InvocationError::Rejected(committed)) => { - return Err(mutation_rejection(committed.output.0, &key_info.table_name)); + Err(error) => return Err(cell_error(error)), + } + } else if no_return { + match self + .client + .command::(&target, mutation_identity()?, Json(input)) + .await + { + Ok(committed) => match committed.output.0 { + ItemMutationOutcome::Applied(_) => None, + _ => { + return Err(StorageError::Internal( + "unexpected successful put result".into(), + )); + } + }, + Err(InvocationError::Rejected(committed)) => { + return Err(mutation_rejection(committed.output.0, &key_info.table_name)); + } + Err(error) => return Err(cell_error(error)), + } + } else { + match self + .client + .command::(&target, mutation_identity()?, Json(input)) + .await + { + Ok(committed) => match committed.output.0 { + ItemMutationOutcome::Applied(_) => None, + _ => { + return Err(StorageError::Internal( + "unexpected successful put result".into(), + )); + } + }, + Err(InvocationError::Rejected(committed)) => { + return Err(mutation_rejection(committed.output.0, &key_info.table_name)); + } + Err(error) => return Err(cell_error(error)), } - Err(error) => return Err(cell_error(error)), }; Ok(if return_old { old } else { None }) }) @@ -162,7 +357,7 @@ impl DataEngine for CellStorage { &partition, &key_info.account_id, Json(PartitionGetInput { - table_id: key_info.table_id, + table_id: key_info.table_id.clone(), epoch, key, }), @@ -180,7 +375,10 @@ impl DataEngine for CellStorage { | PartitionGetOutcome::StaleRoute | PartitionGetOutcome::Sealed | PartitionGetOutcome::NotReady - | PartitionGetOutcome::WrongPartition => Err(stale_partition()), + | PartitionGetOutcome::WrongPartition => { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + Err(stale_partition()) + } }; } let target = target(&key_info.account_id)?; @@ -190,7 +388,7 @@ impl DataEngine for CellStorage { &key_info.account_id, Json(GetItemInput { table_name: key_info.table_name.clone(), - table_id: key_info.table_id, + table_id: key_info.table_id.clone(), key, }), ) @@ -224,20 +422,44 @@ impl DataEngine for CellStorage { let condition = condition.map(|expr| WireCondition::from_core(expr, maps)); Box::pin(async move { if let Some((partition, epoch)) = self.routed_owner(&key_info, &key).await? { + let no_return = !return_old && condition.is_none(); + if no_return { + let dedup_key = crate::item_key(&key, &key_info.base_key_schema) + .map_err(|error| StorageError::Internal(error.to_string()))?; + let result = self + .submit_no_return( + partition, + NoReturnMutation { + table_id: key_info.table_id.clone(), + epoch, + key: dedup_key, + operation: TransactionOperation::Delete(DeleteItemInput { + return_old: false, + table_name: key_info.table_name.clone(), + table_id: key_info.table_id.clone(), + key, + condition: None, + }), + }, + ) + .await; + if result.is_err() { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + } + result?; + return Ok(None); + } + let input = Json(PartitionDeleteInput { + return_old, + table_id: key_info.table_id.clone(), + epoch, + key, + condition, + ttl: false, + }); let outcome = self .client - .command::( - &partition, - mutation_identity()?, - Json(PartitionDeleteInput { - return_old, - table_id: key_info.table_id, - epoch, - key, - condition, - ttl: false, - }), - ) + .command::(&partition, mutation_identity()?, input) .await; let old = match outcome { Ok(committed) => match committed.output.0 { @@ -249,6 +471,7 @@ impl DataEngine for CellStorage { } }, Err(InvocationError::Rejected(committed)) => { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); return Err(partition_delete_rejection(committed.output.0)); } Err(error) => return Err(cell_error(error)), @@ -259,15 +482,21 @@ impl DataEngine for CellStorage { let input = DeleteItemInput { return_old, table_name: key_info.table_name.clone(), - table_id: key_info.table_id, + table_id: key_info.table_id.clone(), key, condition, }; - let old = match self - .client - .command::(&target, mutation_identity()?, Json(input)) - .await - { + let no_return = !return_old && input.condition.is_none(); + let outcome = if no_return { + self.client + .command::(&target, mutation_identity()?, Json(input)) + .await + } else { + self.client + .command::(&target, mutation_identity()?, Json(input)) + .await + }; + let old = match outcome { Ok(committed) => match committed.output.0 { ItemMutationOutcome::Applied(old) => old, _ => { @@ -302,30 +531,80 @@ impl DataEngine for CellStorage { let condition = condition.map(|expr| WireCondition::from_core(expr, maps)); Box::pin(async move { if let Some((partition, epoch)) = self.routed_owner(&key_info, &key).await? { + let no_return = !return_old && !return_new && condition.is_none(); + if no_return { + let dedup_key = crate::item_key(&key, &key_info.base_key_schema) + .map_err(|error| StorageError::Internal(error.to_string()))?; + let result = self + .submit_no_return( + partition, + NoReturnMutation { + table_id: key_info.table_id.clone(), + epoch, + key: dedup_key, + operation: TransactionOperation::Update(UpdateItemInput { + table_name: key_info.table_name.clone(), + table_id: key_info.table_id.clone(), + key, + update, + condition: None, + }), + }, + ) + .await; + if result.is_err() { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + } + result?; + return Ok((None, None)); + } let input = PartitionUpdateInput { - table_id: key_info.table_id, + table_id: key_info.table_id.clone(), epoch, key, update, condition, }; - let outcome = self - .client - .command::(&partition, mutation_identity()?, Json(input)) - .await; - let (old, new) = match outcome { - Ok(committed) => match committed.output.0 { - PartitionUpdateOutcome::Applied { old, new } => (old, new), - _ => { - return Err(StorageError::Internal( - "unexpected partition update".into(), - )); + let (old, new) = if return_old || return_new { + match self + .client + .command::(&partition, mutation_identity()?, Json(input)) + .await + { + Ok(committed) => match committed.output.0 { + PartitionUpdateOutcome::Applied { old, new } => (old, new), + _ => { + return Err(StorageError::Internal( + "unexpected partition update".into(), + )); + } + }, + Err(InvocationError::Rejected(committed)) => { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + return Err(partition_update_rejection(committed.output.0)); } - }, - Err(InvocationError::Rejected(committed)) => { - return Err(partition_update_rejection(committed.output.0)); + Err(error) => return Err(cell_error(error)), + } + } else { + match self + .client + .command::(&partition, mutation_identity()?, Json(input)) + .await + { + Ok(committed) => match committed.output.0 { + PartitionUpdateOutcome::Applied { old, new } => (old, new), + _ => { + return Err(StorageError::Internal( + "unexpected partition update result".into(), + )); + } + }, + Err(InvocationError::Rejected(committed)) => { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + return Err(partition_update_rejection(committed.output.0)); + } + Err(error) => return Err(cell_error(error)), } - Err(error) => return Err(cell_error(error)), }; return Ok(( if return_old { old } else { None }, @@ -333,30 +612,71 @@ impl DataEngine for CellStorage { )); } let target = target(&key_info.account_id)?; + let no_return = !return_old && !return_new && condition.is_none(); let input = UpdateItemInput { table_name: key_info.table_name.clone(), - table_id: key_info.table_id, + table_id: key_info.table_id.clone(), key, update, condition, }; - let (old, new) = match self - .client - .command::(&target, mutation_identity()?, Json(input)) - .await - { - Ok(committed) => match committed.output.0 { - UpdateItemOutcome::Applied { old, new } => (old, new), - _ => { - return Err(StorageError::Internal( - "unexpected successful update result".into(), - )); + let (old, new) = if return_old || return_new { + match self + .client + .command::(&target, mutation_identity()?, Json(input)) + .await + { + Ok(committed) => match committed.output.0 { + UpdateItemOutcome::Applied { old, new } => (old, new), + _ => { + return Err(StorageError::Internal( + "unexpected successful update result".into(), + )); + } + }, + Err(InvocationError::Rejected(committed)) => { + return Err(update_rejection(committed.output.0, &key_info.table_name)); } - }, - Err(InvocationError::Rejected(committed)) => { - return Err(update_rejection(committed.output.0, &key_info.table_name)); + Err(error) => return Err(cell_error(error)), + } + } else if no_return { + match self + .client + .command::(&target, mutation_identity()?, Json(input)) + .await + { + Ok(committed) => match committed.output.0 { + UpdateItemOutcome::AppliedNoReturn => (None, Item::new()), + _ => { + return Err(StorageError::Internal( + "unexpected successful update result".into(), + )); + } + }, + Err(InvocationError::Rejected(committed)) => { + return Err(update_rejection(committed.output.0, &key_info.table_name)); + } + Err(error) => return Err(cell_error(error)), + } + } else { + match self + .client + .command::(&target, mutation_identity()?, Json(input)) + .await + { + Ok(committed) => match committed.output.0 { + UpdateItemOutcome::Applied { old, new } => (old, new), + _ => { + return Err(StorageError::Internal( + "unexpected successful update result".into(), + )); + } + }, + Err(InvocationError::Rejected(committed)) => { + return Err(update_rejection(committed.output.0, &key_info.table_name)); + } + Err(error) => return Err(cell_error(error)), } - Err(error) => return Err(cell_error(error)), }; Ok(( if return_old { old } else { None }, @@ -485,7 +805,10 @@ impl DataEngine for CellStorage { | PartitionQueryOutcome::StaleRoute | PartitionQueryOutcome::Sealed | PartitionQueryOutcome::NotReady - | PartitionQueryOutcome::WrongPartition => Err(stale_partition()), + | PartitionQueryOutcome::WrongPartition => { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + Err(stale_partition()) + } }; } if let Some(start) = exclusive_start_key { @@ -532,56 +855,69 @@ impl DataEngine for CellStorage { }) }) .transpose()?; - if let Some((items, last_evaluated_key)) = self - .scan_routed( - &key_info, - limit, - exclusive_start_key.clone(), - segment, - index_name.as_deref(), - ) - .await? - { - // Retain the unfiltered cursor so empty segment pages still advance. - return Ok(( - scan_segment(items, &key_info, segment, index_name.as_deref())?, - last_evaluated_key, - )); - } - let target = target(&key_info.account_id)?; - let output = self - .query_resolving::( - &target, - &key_info.account_id, - Json(ScanItemsInput { - index_name: index_name.clone(), - table_name: key_info.table_name.clone(), - table_id: key_info.table_id.clone(), + let mut cursor = exclusive_start_key; + loop { + let (items, next) = if let Some(page) = self + .scan_routed( + &key_info, limit, - exclusive_start_key, - }), - ) - .await?; - match output.output.0 { - ScanItemsOutcome::Page { - items, - last_evaluated_key, - } => Ok(( - scan_segment(items, &key_info, segment, index_name.as_deref())?, - last_evaluated_key, - )), - ScanItemsOutcome::Conflict(_) => Err(StorageError::Transient( - "scan range is locked by a transaction".into(), - )), - ScanItemsOutcome::TableNotFound => { - Err(StorageError::TableNotFound(key_info.table_name)) + cursor.clone(), + segment, + index_name.as_deref(), + ) + .await? + { + page + } else { + let account = target(&key_info.account_id)?; + let output = self + .query_resolving::( + &account, + &key_info.account_id, + Json(ScanItemsInput { + index_name: index_name.clone(), + table_name: key_info.table_name.clone(), + table_id: key_info.table_id.clone(), + limit, + exclusive_start_key: cursor.clone(), + }), + ) + .await?; + match output.output.0 { + ScanItemsOutcome::Page { + items, + last_evaluated_key, + } => (items, last_evaluated_key), + ScanItemsOutcome::Conflict(_) => { + return Err(StorageError::Transient( + "scan range is locked by a transaction".into(), + )); + } + ScanItemsOutcome::TableNotFound => { + return Err(StorageError::TableNotFound(key_info.table_name)); + } + ScanItemsOutcome::InvalidKey => { + return Err(StorageError::Validation( + "scan continuation key does not match table schema".into(), + )); + } + ScanItemsOutcome::InvalidLimit => { + return Err(StorageError::Validation( + "scan limit must be positive".into(), + )); + } + } + }; + let visible = scan_segment(items, &key_info, segment, index_name.as_deref())?; + if !visible.is_empty() || next.is_none() { + return Ok((visible, next)); } - ScanItemsOutcome::InvalidKey => Err(StorageError::Validation( - "scan continuation key does not match table schema".into(), - )), - ScanItemsOutcome::InvalidLimit => Err(StorageError::Validation( - "scan limit must be positive".into(), - )), + if cursor == next { + return Err(StorageError::Internal( + "scan continuation did not advance".into(), + )); + } + cursor = next; } }) } @@ -597,6 +933,12 @@ impl DataEngine for CellStorage { .collect(); Box::pin(async move { let (account_id, inputs) = prepared?; + if let Some(items) = self + .try_local_transaction_read(&account_id, inputs.clone(), routing.clone()) + .await? + { + return validate_read_size(items); + } // Saved participant images keep a legal aggregate read from crossing // one Cell response. Use the same serialization boundary for every route. self.transaction_read(&account_id, inputs, routing).await @@ -659,6 +1001,19 @@ impl DataEngine for CellStorage { )); } let count = operations.len(); + if token.is_none() + && self + .try_local_transaction_write( + &account_id, + operations.clone(), + routing.clone(), + count, + &return_old_on_failure, + ) + .await? + { + return Ok(()); + } let admitted = self .admit_transaction(&account_id, token, operations, routing) .await?; @@ -734,6 +1089,130 @@ fn segment_bounds(segment: u64, total: u64) -> ([u8; 16], Option<[u8; 16]>) { ) } +impl CellStorage { + async fn try_local_transaction_write( + &self, + account_id: &str, + operations: Vec, + routing: Vec<(TableKeyInfo, Item)>, + count: usize, + return_old_on_failure: &[bool], + ) -> Result { + let participants = self + .route_transaction_participants(account_id, operations.clone(), routing) + .await?; + if participants.len() != 1 { + return Ok(false); + } + let Some((_, participant)) = participants.into_iter().next() else { + return Ok(false); + }; + // A coordinator is required as soon as a request spans Cells. The local + // commands already stage and apply every operation in one SQLite + // transaction, so bypassing the coordinator is safe for this case. + let mut participant_operations = participant.operations; + participant_operations.sort_unstable_by_key(|operation| operation.index); + let participant_operations = participant_operations + .into_iter() + .map(|operation| operation.operation) + .collect::>(); + if participant_operations.len() != operations.len() { + return Ok(false); + } + match participant.target { + crate::CoordinatorParticipantTarget::Account => { + let target = target(account_id)?; + let result = if return_old_on_failure.iter().all(|return_old| !return_old) { + self.client + .command::( + &target, + mutation_identity()?, + Json(TransactWriteInput { + operations: participant_operations.clone(), + }), + ) + .await + } else { + self.client + .command::( + &target, + mutation_identity()?, + Json(TransactWriteInput { + operations: participant_operations, + }), + ) + .await + }; + match result { + Ok(committed) => match committed.output.0 { + TransactionOutcome::Applied => Ok(true), + TransactionOutcome::Rejected { index, reason } => Err( + transaction_canceled(index, reason, count, return_old_on_failure), + ), + }, + Err(InvocationError::Rejected(committed)) => match committed.output.0 { + TransactionOutcome::Applied => Err(StorageError::Internal( + "unexpected rejected local account transaction".into(), + )), + TransactionOutcome::Rejected { index, reason } => Err( + transaction_canceled(index, reason, count, return_old_on_failure), + ), + }, + Err(error) => Err(cell_error(error)), + } + } + crate::CoordinatorParticipantTarget::Data { + table_id, + partition_id, + epoch, + } => { + let target = crate::data_target(account_id, &table_id, &partition_id) + .map_err(|error| StorageError::Internal(error.to_string()))?; + let result = if return_old_on_failure.iter().all(|return_old| !return_old) { + self.client + .command::( + &target, + mutation_identity()?, + Json(PartitionTransactWriteInput { + table_id, + epoch, + operations: participant_operations.clone(), + }), + ) + .await + } else { + self.client + .command::( + &target, + mutation_identity()?, + Json(PartitionTransactWriteInput { + table_id, + epoch, + operations: participant_operations, + }), + ) + .await + }; + match result { + Ok(committed) => local_partition_transaction_result( + committed.output.0, + count, + return_old_on_failure, + ), + Err(InvocationError::Rejected(committed)) => { + local_partition_transaction_result( + committed.output.0, + count, + return_old_on_failure, + ) + } + Err(error) => Err(cell_error(error)), + } + } + } + } +} + pub(super) fn transaction_canceled( index: usize, reason: TransactionFailure, @@ -769,6 +1248,44 @@ pub(super) fn transaction_canceled( StorageError::TransactionCanceled(reasons) } +fn local_partition_transaction_result( + outcome: PartitionTransactWriteOutcome, + count: usize, + return_old_on_failure: &[bool], +) -> Result { + match outcome { + PartitionTransactWriteOutcome::Applied => Ok(true), + PartitionTransactWriteOutcome::Rejected { index, reason } => Err(transaction_canceled( + index, + reason, + count, + return_old_on_failure, + )), + PartitionTransactWriteOutcome::NotInstalled + | PartitionTransactWriteOutcome::StaleRoute + | PartitionTransactWriteOutcome::Sealed + | PartitionTransactWriteOutcome::NotReady + | PartitionTransactWriteOutcome::WrongPartition => Err(stale_partition()), + } +} + +fn local_partition_transaction_read_result( + outcome: PartitionTransactReadOutcome, + count: usize, +) -> Result>>, StorageError> { + match outcome { + PartitionTransactReadOutcome::Applied(items) => Ok(Some(items)), + PartitionTransactReadOutcome::Rejected { index, reason } => { + Err(transaction_canceled(index, reason, count, &[])) + } + PartitionTransactReadOutcome::NotInstalled + | PartitionTransactReadOutcome::StaleRoute + | PartitionTransactReadOutcome::Sealed + | PartitionTransactReadOutcome::NotReady + | PartitionTransactReadOutcome::WrongPartition => Err(stale_partition()), + } +} + struct PreparedQuery { partition_key: Item, sort: Option, @@ -912,7 +1429,7 @@ fn update_rejection(outcome: UpdateItemOutcome, table_name: &str) -> StorageErro } UpdateItemOutcome::ConditionFailed(old) => StorageError::ConditionFailed(old), UpdateItemOutcome::InvalidExpression(message) => StorageError::Validation(message), - UpdateItemOutcome::Applied { .. } => { + UpdateItemOutcome::Applied { .. } | UpdateItemOutcome::AppliedNoReturn => { StorageError::Internal("unexpected rejected update result".into()) } } diff --git a/src/backend/data/routed.rs b/src/backend/data/routed.rs index d896175..823ca00 100644 --- a/src/backend/data/routed.rs +++ b/src/backend/data/routed.rs @@ -35,17 +35,17 @@ impl CellStorage { // Placement is immutable for a table generation. Once the account Cell // has confirmed an account-local table, there can never be a directory // for this table ID, so skip the route and placement reads on hot paths. - if !key_info.table_id.is_empty() - && self - .account_placement_cache - .read() - .await - .contains(&key_info.table_id) - { + if !key_info.table_id.is_empty() && self.account_placement_cached(&key_info.table_id) { return Ok(None); } let hash = data_key_hash(&key_info.table_id, key, &key_info.base_key_schema) .map_err(|error| StorageError::Validation(error.to_string()))?; + let cache_key = (key_info.account_id.clone(), key_info.table_id.clone()); + if self.route_cache_enabled + && let Some(partition) = self.cached_route(&cache_key, hash) + { + return Ok(Some(partition)); + } let account = target(&key_info.account_id)?; match crate::read_route_page( &self.client, @@ -62,17 +62,25 @@ impl CellStorage { RoutePageOutcome::Unrouted => { self.require_account_placement(key_info).await?; if !key_info.table_id.is_empty() { - self.account_placement_cache - .write() - .await - .insert(key_info.table_id.clone()); + self.cache_account_placement(key_info.table_id.clone()); } Ok(None) } - RoutePageOutcome::Changed => Err(stale_partition()), - RoutePageOutcome::Page { partitions, .. } => { + RoutePageOutcome::Changed => { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + Err(stale_partition()) + } + RoutePageOutcome::Page { + epoch: _, + partitions, + has_more, + } => { let range = partitions.first().ok_or_else(stale_partition)?; - Ok(Some((range.partition_id, range.epoch))) + let selected = (range.partition_id, range.epoch); + if self.route_cache_enabled { + self.cache_route(cache_key, partitions, !has_more); + } + Ok(Some(selected)) } } } @@ -355,7 +363,7 @@ pub(super) fn partition_update_rejection(outcome: PartitionUpdateOutcome) -> Sto | PartitionUpdateOutcome::Sealed | PartitionUpdateOutcome::NotReady | PartitionUpdateOutcome::WrongPartition => stale_partition(), - PartitionUpdateOutcome::Applied { .. } => { + PartitionUpdateOutcome::Applied { .. } | PartitionUpdateOutcome::AppliedNoReturn => { StorageError::Internal("unexpected rejected partition update".into()) } } diff --git a/src/backend/global_index.rs b/src/backend/global_index.rs index 677251b..40ab8ba 100644 --- a/src/backend/global_index.rs +++ b/src/backend/global_index.rs @@ -26,6 +26,46 @@ impl CellStorage { source: &CellTarget, table_id: &str, ) -> Result { + self.project_index_changes_bounded(account_id, source, table_id, 1) + .await + } + + // A background pass may advance several distinct journal entries. A failed + // index keeps its entry for replay, but cannot hold newer healthy-index + // versions behind repeated attempts at the same entry in one pass. + async fn project_index_changes_bounded( + &self, + account_id: &str, + source: &CellTarget, + table_id: &str, + max_entries: usize, + ) -> Result { + let mut seen = std::collections::HashSet::new(); + let mut advanced = false; + let mut failure = None; + for _ in 0..max_entries { + let Some((id, error)) = self + .project_index_change_once(account_id, source, table_id, &seen) + .await? + else { + break; + }; + seen.insert(id); + advanced = true; + if let Some(error) = error { + failure.get_or_insert(error); + } + } + failure.map_or(Ok(advanced), Err) + } + + async fn project_index_change_once( + &self, + account_id: &str, + source: &CellTarget, + table_id: &str, + seen: &std::collections::HashSet<[u8; 32]>, + ) -> Result)>, StorageError> { let account = target(account_id)?; if source.tenant() != account.tenant() || source.application() != account.application() @@ -48,8 +88,11 @@ impl CellStorage { .output .0; let Some(header) = header else { - return Ok(false); + return Ok(None); }; + if seen.contains(&header.id) { + return Ok(None); + } let mut bytes = Vec::new(); while bytes.len() < header.bytes as usize { let input = Json(IndexChangeChunk { @@ -70,7 +113,7 @@ impl CellStorage { .0; // Another worker can acknowledge only after the whole entry projects. let Some(part) = part else { - return Ok(true); + return Ok(Some((header.id, None))); }; if part.is_empty() || bytes.len() + part.len() > header.bytes as usize { return Err(StorageError::Internal( @@ -127,7 +170,7 @@ impl CellStorage { .await } .map_err(cell_error)?; - failure.map_or(Ok(true), Err) + Ok(Some((header.id, failure))) } async fn project_to_index( @@ -451,7 +494,7 @@ impl CellStorage { .map(|source| async move { provisioner.recover_projection_owner(&source, nodes).await?; if project { - self.project_index_changes(account, &source, table_id) + self.project_index_changes_bounded(account, &source, table_id, 4) .await .map(|_| ()) } else { diff --git a/src/backend/metadata_cache.rs b/src/backend/metadata_cache.rs new file mode 100644 index 0000000..22db02f --- /dev/null +++ b/src/backend/metadata_cache.rs @@ -0,0 +1,129 @@ +//! Short-lived metadata responses for the explicitly stale-tolerant serving mode. + +use std::{ + collections::HashMap, + sync::{RwLock, RwLockReadGuard, RwLockWriteGuard}, + time::{Duration, Instant}, +}; + +use extenddb_core::types::{ListTablesOutput, TableDescription}; + +const TTL: Duration = Duration::from_millis(500); +const MAX_ENTRIES: usize = 128; + +#[derive(Default)] +pub(super) struct MetadataCache { + state: RwLock, +} + +#[derive(Default)] +struct State { + generation: u64, + descriptions: HashMap<(String, String), Entry>, + listings: HashMap<(String, i64, Option), Entry>, +} + +struct Entry { + at: Instant, + generation: u64, + value: T, +} + +impl MetadataCache { + fn read(&self) -> RwLockReadGuard<'_, State> { + match self.state.read() { + Ok(state) => state, + Err(poisoned) => poisoned.into_inner(), + } + } + + fn write(&self) -> RwLockWriteGuard<'_, State> { + match self.state.write() { + Ok(state) => state, + Err(poisoned) => poisoned.into_inner(), + } + } + + pub(super) fn generation(&self) -> u64 { + self.read().generation + } + + pub(super) fn description(&self, account_id: &str, name: &str) -> Option { + let state = self.read(); + let entry = state + .descriptions + .get(&(account_id.to_owned(), name.to_owned()))?; + (entry.at.elapsed() < TTL && entry.generation == state.generation) + .then(|| entry.value.clone()) + } + + pub(super) fn listing( + &self, + account_id: &str, + limit: i64, + start: &Option, + ) -> Option { + let state = self.read(); + let entry = state + .listings + .get(&(account_id.to_owned(), limit, start.clone()))?; + (entry.at.elapsed() < TTL && entry.generation == state.generation) + .then(|| entry.value.clone()) + } + + pub(super) fn insert_description( + &self, + account_id: &str, + name: String, + generation: u64, + value: TableDescription, + ) { + let mut state = self.write(); + if state.generation != generation { + return; + } + if state.descriptions.len() >= MAX_ENTRIES { + state.descriptions.clear(); + } + state.descriptions.insert( + (account_id.to_owned(), name), + Entry { + at: Instant::now(), + generation, + value, + }, + ); + } + + pub(super) fn insert_listing( + &self, + account_id: &str, + limit: i64, + start: Option, + generation: u64, + value: ListTablesOutput, + ) { + let mut state = self.write(); + if state.generation != generation { + return; + } + if state.listings.len() >= MAX_ENTRIES { + state.listings.clear(); + } + state.listings.insert( + (account_id.to_owned(), limit, start), + Entry { + at: Instant::now(), + generation, + value, + }, + ); + } + + pub(super) fn invalidate(&self) { + let mut state = self.write(); + state.generation = state.generation.saturating_add(1); + state.descriptions.clear(); + state.listings.clear(); + } +} diff --git a/src/backend/recovery.rs b/src/backend/recovery.rs index 977f4ca..d2d1467 100644 --- a/src/backend/recovery.rs +++ b/src/backend/recovery.rs @@ -4,20 +4,22 @@ use cellule_runtime::client::{InvocationError, Observed, Receipt}; use cellule_runtime::identity::CellTarget; use extenddb_storage::error::StorageError; use futures_util::{StreamExt, stream}; +use std::time::Duration; use super::{CellStorage, cell_error, mutation_identity}; use crate::{ CoordinatorDecision, CoordinatorParticipantTarget, CoordinatorPhaseInput, - CoordinatorPhaseOutcome, DecideCrossCellTransaction, DecideCrossCellTransactionInput, - DecideCrossCellTransactionOutcome, Json, NAMESPACE, ParticipantTransactionState, - PendingCrossCellTransaction, PendingTransactionCursor, PendingTransactionState, - ReadAccountTransaction, ReadCrossCellTransaction, ReadCrossCellTransactionInput, - ReadPartitionTransaction, ReadPendingCrossCellTransactions, - ReadPendingCrossCellTransactionsInput, ReadPendingTransactionBoundary, ReadTransactionInput, - ReadUnresolvedCoordinatorParticipants, RecordParticipantResolution, RecordReadResultRelease, - ReleaseAccountTransactionReads, ReleasePartitionTransactionReads, ResolveAccountTransaction, - ResolvePartitionTransaction, ResolveTransactionInput, ResolveTransactionOutcome, - UnresolvedCoordinatorParticipant, account_target, coordinator_target, data_target, + CoordinatorPhaseOutcome, CrossCellTransactionStatus, DecideCrossCellTransaction, + DecideCrossCellTransactionInput, DecideCrossCellTransactionOutcome, Json, NAMESPACE, + ParticipantTransactionState, PendingCrossCellTransaction, PendingTransactionCursor, + PendingTransactionState, ReadAccountTransaction, ReadCrossCellTransaction, + ReadCrossCellTransactionInput, ReadPartitionTransaction, ReadPendingCrossCellTransactions, + ReadPendingCrossCellTransactionsInput, ReadPendingTransactionBoundary, ReadResultRelease, + ReadTransactionInput, ReadUnresolvedCoordinatorParticipants, RecordParticipantResolutions, + RecordReadResultReleases, RecordReadResultReleasesInput, ReleaseAccountTransactionReads, + ReleasePartitionTransactionReads, ResolveAccountTransaction, ResolvePartitionTransaction, + ResolveTransactionInput, ResolveTransactionOutcome, UnresolvedCoordinatorParticipant, + account_target, coordinator_target, data_target, }; impl CellStorage { @@ -172,6 +174,16 @@ impl CellStorage { .output .0 .ok_or_else(|| StorageError::Internal("coordinator transaction is missing".into()))?; + self.finish_decided_cross_cell_transaction_from_status(&coordinator, &read, status) + .await + } + + pub(super) async fn finish_decided_cross_cell_transaction_from_status( + &self, + coordinator: &CellTarget, + read: &ReadCrossCellTransactionInput, + status: CrossCellTransactionStatus, + ) -> Result<(), StorageError> { let commit = match status.decision { CoordinatorDecision::Begin => { return Err(StorageError::Transient( @@ -187,7 +199,7 @@ impl CellStorage { } let participants = self .client - .query::(&coordinator, None, Json(read.clone())) + .query::(coordinator, None, Json(read.clone())) .await .map_err(cell_error)? .output @@ -197,19 +209,58 @@ impl CellStorage { // window so a slow owner cannot hold healthy keys, without fanning one // request out to all 100 participants or detaching work on cancellation. let mut resolving = stream::iter(participants) - .map(|participant| self.finish_participant(&coordinator, &read, participant, commit)) + .map(|participant| self.finish_participant(coordinator, read, participant, commit)) .buffer_unordered(4); - while let Some(result) = resolving.next().await { - if let Err(error) = result { - failure.get_or_insert(error); + let mut read_releases = Vec::new(); + let mut resolutions = Vec::new(); + loop { + // Publish completed participants even if another owner has not + // replied. A short quiet window still coalesces nearby receipts. + let next = if read_releases.is_empty() && resolutions.is_empty() { + resolving.next().await + } else { + match tokio::time::timeout(Duration::from_millis(2), resolving.next()).await { + Ok(next) => next, + Err(_) => { + self.record_resolution_progress( + coordinator, + read, + &mut read_releases, + &mut resolutions, + ) + .await?; + continue; + } + } + }; + let Some(result) = next else { + break; + }; + match result { + Ok((true, input)) => read_releases.push(input), + Ok((false, input)) => resolutions.push(input), + Err(error) => { + failure.get_or_insert(error); + } + } + if read_releases.len() + resolutions.len() >= 16 { + self.record_resolution_progress( + coordinator, + read, + &mut read_releases, + &mut resolutions, + ) + .await?; } } + self.record_resolution_progress(coordinator, read, &mut read_releases, &mut resolutions) + .await?; if let Some(error) = failure { return Err(error); } let final_status = self .client - .query::(&coordinator, None, Json(read.clone())) + .query::(coordinator, None, Json(read.clone())) .await .map_err(cell_error)? .output @@ -225,13 +276,78 @@ impl CellStorage { Ok(()) } + async fn record_resolution_progress( + &self, + coordinator: &CellTarget, + read: &ReadCrossCellTransactionInput, + read_releases: &mut Vec, + resolutions: &mut Vec, + ) -> Result<(), StorageError> { + if !read_releases.is_empty() { + let receipts = read_releases + .drain(..) + .map(|input| ReadResultRelease { + position: input.position, + participant_cell: input.participant_cell, + sequence: input.sequence, + }) + .collect(); + let recorded = self + .client + .command::( + coordinator, + mutation_identity()?, + Json(RecordReadResultReleasesInput { + account_id: read.account_id.clone(), + transaction_id: read.transaction_id, + routing_key: read.routing_key.clone(), + releases: receipts, + }), + ) + .await + .map_err(cell_error)?; + if recorded.output.0.iter().any(|outcome| { + !matches!( + outcome, + CoordinatorPhaseOutcome::Recorded | CoordinatorPhaseOutcome::Replay + ) + }) { + return Err(StorageError::Internal( + "coordinator rejected read result release".into(), + )); + } + } + if !resolutions.is_empty() { + let recorded = self + .client + .command::( + coordinator, + mutation_identity()?, + Json(std::mem::take(resolutions)), + ) + .await + .map_err(cell_error)?; + if recorded.output.0.iter().any(|outcome| { + !matches!( + outcome, + CoordinatorPhaseOutcome::Recorded | CoordinatorPhaseOutcome::Replay + ) + }) { + return Err(StorageError::Internal( + "coordinator rejected participant resolution".into(), + )); + } + } + Ok(()) + } + async fn finish_participant( &self, coordinator: &CellTarget, read: &ReadCrossCellTransactionInput, participant: UnresolvedCoordinatorParticipant, commit: bool, - ) -> Result<(), StorageError> { + ) -> Result<(bool, CoordinatorPhaseInput), StorageError> { let position = participant.position; let target = match participant.target { CoordinatorParticipantTarget::Account => account_target(&read.account_id) @@ -277,30 +393,7 @@ impl CellStorage { participant_cell: *target.cell_id().as_bytes(), sequence: receipt.commit_sequence, }); - let identity = mutation_identity()?; - let recorded = if participant.release_read_result { - self.client - .command::(coordinator, identity, input) - .await - } else { - self.client - .command::(coordinator, identity, input) - .await - }; - match recorded { - Ok(committed) - if matches!( - committed.output.0, - CoordinatorPhaseOutcome::Recorded | CoordinatorPhaseOutcome::Replay - ) => - { - Ok(()) - } - Ok(_) | Err(InvocationError::Rejected(_)) => Err(StorageError::Internal( - "coordinator rejected participant resolution".into(), - )), - Err(error) => Err(cell_error(error)), - } + Ok((participant.release_read_result, input.0)) } async fn resolve_participant( @@ -314,19 +407,6 @@ impl CellStorage { transaction_id, coordinator_cell: *coordinator.cell_id().as_bytes(), }; - let observed = self.participant_state(target, input.clone()).await?; - match (commit, observed.output.0) { - (true, ParticipantTransactionState::Committed) - | (false, ParticipantTransactionState::Aborted) => return Ok(observed.receipt), - (true, ParticipantTransactionState::Prepared) - | (false, ParticipantTransactionState::Prepared) - | (false, ParticipantTransactionState::Missing) => {} - _ => { - return Err(StorageError::Internal( - "participant state contradicts coordinator decision".into(), - )); - } - } let resolve = Json(ResolveTransactionInput { transaction_id, coordinator_cell: input.coordinator_cell, diff --git a/src/backend/transaction.rs b/src/backend/transaction.rs index cc415a5..e3fe27f 100644 --- a/src/backend/transaction.rs +++ b/src/backend/transaction.rs @@ -4,6 +4,7 @@ use cellule_runtime::client::{InvocationError, Receipt}; use cellule_runtime::identity::CellTarget; use extenddb_storage::error::StorageError; use futures_util::{StreamExt, stream}; +use std::time::Duration; use super::transaction_transport::PhaseError; use super::{CellStorage, cell_error, mutation_identity}; @@ -14,8 +15,8 @@ use crate::{ ParticipantTransactionState, PrepareAccountTransaction, PrepareAccountTransactionInput, PreparePartitionTransaction, PreparePartitionTransactionInput, PrepareTransactionOutcome, ReadCoordinatorParticipantInput, ReadCrossCellTransaction, ReadCrossCellTransactionInput, - ReadTransactionInput, ReadUnresolvedCoordinatorParticipants, RecordParticipantPrepare, - TransactionFailure, account_target, coordinator_target, data_target, + ReadTransactionInput, ReadUnresolvedCoordinatorParticipants, RecordParticipantPrepares, + TransactionCommandInput, TransactionFailure, account_target, coordinator_target, data_target, }; impl CellStorage { @@ -30,6 +31,18 @@ impl CellStorage { routing_key: &[u8], transaction_id: [u8; 16], ) -> Result { + Ok(self + .resume_cross_cell_transaction_status(account_id, routing_key, transaction_id) + .await? + .decision) + } + + pub(super) async fn resume_cross_cell_transaction_status( + &self, + account_id: &str, + routing_key: &[u8], + transaction_id: [u8; 16], + ) -> Result { let coordinator = coordinator_target(account_id, routing_key) .map_err(|error| StorageError::Internal(error.to_string()))?; let read = ReadCrossCellTransactionInput { @@ -126,7 +139,21 @@ impl CellStorage { // capacity outcomes retain the prior participant order. let mut prepared = stream::iter(attempts.into_iter().map( |(position, payload, target, input)| async move { - let result = self.prepare_transaction_participant(&target, input).await; + let mut admission_retries = 0; + let result = loop { + let result = self + .prepare_transaction_participant(&target, input.clone()) + .await; + match result { + Err(PhaseError::Capacity(cellule_runtime::Error::Capacity(_))) + if admission_retries < 4 => + { + admission_retries += 1; + tokio::time::sleep(Duration::from_millis(2 << admission_retries)).await; + } + other => break other, + } + }; (position, payload, target, result) }, )) @@ -193,54 +220,51 @@ impl CellStorage { evidence.push((position, target, receipt)); } - // Evidence records are independent CAS updates on the coordinator. - // Publish them concurrently after all participant prepares succeed; - // durable ordering is carried by each participant position. - let recorded = stream::iter(evidence.into_iter().map(|(position, target, receipt)| { - let coordinator = coordinator.clone(); - let read = read.clone(); - async move { - let result = self - .client - .command::( - &coordinator, - mutation_identity()?, - Json(CoordinatorPhaseInput { - account_id: read.account_id, - transaction_id, - routing_key: read.routing_key, - position, - participant_cell: *target.cell_id().as_bytes(), - sequence: receipt.commit_sequence, - }), + // Participant prepares already ran concurrently. Record their receipts + // in one coordinator command so the coordinator publishes one durable + // evidence update instead of one round trip per participant. + let phases: Vec<_> = evidence + .into_iter() + .map(|(position, target, receipt)| CoordinatorPhaseInput { + account_id: read.account_id.clone(), + transaction_id, + routing_key: read.routing_key.clone(), + position, + participant_cell: *target.cell_id().as_bytes(), + sequence: receipt.commit_sequence, + }) + .collect(); + if phases.is_empty() { + return self + .decide_transaction(&coordinator, &read, CoordinatorDecision::Commit) + .await; + } + let recorded = self + .client + .command::(&coordinator, mutation_identity()?, Json(phases)) + .await; + match recorded { + Ok(result) + if result.output.0.iter().all(|outcome| { + matches!( + outcome, + CoordinatorPhaseOutcome::Recorded | CoordinatorPhaseOutcome::Replay ) - .await; - Ok::<_, StorageError>(result) + }) => {} + Ok(result) + if result + .output + .0 + .contains(&CoordinatorPhaseOutcome::WrongDecision) => + { + return self.finish_transaction(&coordinator, &read).await; } - })) - .buffer_unordered(8) - .collect::>() - .await; - for result in recorded { - let recorded = result?; - match recorded { - Ok(result) - if matches!( - result.output.0, - CoordinatorPhaseOutcome::Recorded | CoordinatorPhaseOutcome::Replay - ) => {} - Err(InvocationError::Rejected(result)) - if result.output.0 == CoordinatorPhaseOutcome::WrongDecision => - { - return self.finish_transaction(&coordinator, &read).await; - } - Ok(_) | Err(InvocationError::Rejected(_)) => { - return Err(StorageError::Internal( - "coordinator refused prepare evidence".into(), - )); - } - Err(error) => return Err(cell_error(error)), + Ok(_) | Err(InvocationError::Rejected(_)) => { + return Err(StorageError::Internal( + "coordinator refused prepare evidence".into(), + )); } + Err(error) => return Err(cell_error(error)), } self.decide_transaction(&coordinator, &read, CoordinatorDecision::Commit) .await @@ -264,20 +288,16 @@ impl CellStorage { &self, coordinator: &CellTarget, read: &ReadCrossCellTransactionInput, - ) -> Result { + ) -> Result { let status = self.transaction_status(coordinator, read).await?; if status.decision == CoordinatorDecision::Begin { return Err(StorageError::Transient( "transaction decision remains pending".into(), )); } - self.finish_decided_cross_cell_transaction( - &read.account_id, - &read.routing_key, - read.transaction_id, - ) - .await?; - Ok(status.decision) + self.finish_decided_cross_cell_transaction_from_status(coordinator, read, status.clone()) + .await?; + Ok(status) } async fn decide_transaction( @@ -285,7 +305,7 @@ impl CellStorage { coordinator: &CellTarget, read: &ReadCrossCellTransactionInput, decision: CoordinatorDecision, - ) -> Result { + ) -> Result { let result = self .client .command::( @@ -338,22 +358,57 @@ impl CellStorage { // the participant Cell. Avoid a separate state query on the normal // first-attempt path; only an ambiguous reply needs a follow-up read. let identity = mutation_identity()?; + let inline = match &input { + ParticipantPrepare::Account(input) => serde_json::to_vec(input), + ParticipantPrepare::Data(input) => serde_json::to_vec(input), + } + .map_err(|error| StorageError::Internal(error.to_string()))? + .len() + <= crate::transaction_transport::INLINE_BYTES; let result = match input { ParticipantPrepare::Account(input) => { - let reference = self - .upload_transaction::(target, identity, &input) - .await?; - self.client - .command::(target, identity, Json(reference)) - .await + if inline { + self.client + .command::( + target, + identity, + Json(TransactionCommandInput::Inline(input)), + ) + .await + } else { + let reference = self + .upload_transaction::(target, identity, &input) + .await?; + self.client + .command::( + target, + identity, + Json(TransactionCommandInput::Reference(reference)), + ) + .await + } } ParticipantPrepare::Data(input) => { - let reference = self - .upload_transaction::(target, identity, &input) - .await?; - self.client - .command::(target, identity, Json(reference)) - .await + if inline { + self.client + .command::( + target, + identity, + Json(TransactionCommandInput::Inline(input)), + ) + .await + } else { + let reference = self + .upload_transaction::(target, identity, &input) + .await?; + self.client + .command::( + target, + identity, + Json(TransactionCommandInput::Reference(reference)), + ) + .await + } } }; match result { @@ -382,6 +437,7 @@ impl CellStorage { } } +#[derive(Clone)] enum ParticipantPrepare { Account(PrepareAccountTransactionInput), Data(PreparePartitionTransactionInput), diff --git a/src/backend/transaction_read.rs b/src/backend/transaction_read.rs index f06d782..0e583da 100644 --- a/src/backend/transaction_read.rs +++ b/src/backend/transaction_read.rs @@ -1,16 +1,21 @@ //! Serializable transactional reads from immutable participant snapshots. +use cellule_runtime::client::InvocationError; use extenddb_core::types::{Item, TableKeyInfo}; use extenddb_storage::error::StorageError; use futures_util::{StreamExt, stream}; +use std::collections::HashMap; +use std::time::Duration; use super::{CellStorage, cell_error, mutation_identity}; use crate::{ - BeginReadResultRelease, CoordinatorDecision, CoordinatorParticipantTarget, GetItemInput, Json, - ReadAccountTransactionResult, ReadCoordinatorParticipantInput, ReadCrossCellTransaction, - ReadPartitionTransactionResult, ReadTransactionInput, ReadTransactionResultInput, - TransactionFailure, TransactionOperation, TransactionReadResult, account_target, - coordinator_target, data_target, + BeginReadResultRelease, CoordinatorDecision, CoordinatorParticipantTarget, + CoordinatorPhaseOutcome, GetItemInput, Json, ReadAccountTransactionResult, + ReadCoordinatorParticipantInput, ReadPartitionTransactionResult, ReadResultRelease, + ReadTransactionInput, ReadTransactionResultInput, RecordReadResultReleases, + RecordReadResultReleasesInput, ReleaseAccountTransactionReads, + ReleasePartitionTransactionReads, TransactionFailure, TransactionOperation, + TransactionReadResult, account_target, coordinator_target, data_target, }; #[derive(Clone)] @@ -50,31 +55,27 @@ impl CellStorage { let identity = admitted.identity; let coordinator = coordinator_target(account_id, &identity.routing_key) .map_err(|error| StorageError::Internal(error.to_string()))?; - let status = self - .client - .query::(&coordinator, None, Json(identity.clone())) - .await - .map_err(cell_error)? - .output - .0 - .ok_or_else(|| StorageError::Internal("read transaction disappeared".into()))?; let participant_inputs = - (0..status.participant_count).map(|position| ReadCoordinatorParticipantInput { + (0..admitted.participant_count).map(|position| ReadCoordinatorParticipantInput { account_id: account_id.into(), transaction_id: identity.transaction_id, routing_key: identity.routing_key.clone(), position, chunk: 0, }); - let participant_results = stream::iter( - participant_inputs - .map(|input| async { self.coordinator_participant(&coordinator, input).await }), - ) + let participant_results = stream::iter(participant_inputs.map(|input| async { + let position = input.position; + ( + position, + self.coordinator_participant(&coordinator, input).await, + ) + })) .buffer_unordered(8) .collect::>() .await; - let mut image_reads = Vec::new(); - for result in participant_results { + let mut image_reads: HashMap<[u8; 32], Vec<_>> = HashMap::new(); + let mut participant_targets = Vec::with_capacity(participant_results.len()); + for (participant_position, result) in participant_results { let participant = result?.ok_or_else(|| { StorageError::Internal("committed read operations are missing".into()) })?; @@ -89,13 +90,19 @@ impl CellStorage { } => data_target(account_id, table_id, partition_id).map(ReadTarget::Data), } .map_err(|error| StorageError::Internal(error.to_string()))?; + participant_targets.push((participant_position, target.clone())); for (position, operation) in participant.operations.into_iter().enumerate() { if !matches!(operation.operation, TransactionOperation::Read(_)) { return Err(StorageError::Internal( "read transaction contains a write".into(), )); } - image_reads.push(( + let cell = match &target { + ReadTarget::Account(target) | ReadTarget::Data(target) => { + *target.cell_id().as_bytes() + } + }; + image_reads.entry(cell).or_default().push(( operation.index, target.clone(), Json(ReadTransactionResultInput { @@ -109,8 +116,12 @@ impl CellStorage { )); } } - let images = stream::iter(image_reads.into_iter().map( - |(index, target, input)| async move { + // One image query reserves up to the Cell wire result ceiling. Fetch + // each participant's images serially to stay inside its 16 MiB mailbox, + // while independent participant Cells can still make progress together. + let images = stream::iter(image_reads.into_values().map(|reads| async move { + let mut group = Vec::with_capacity(reads.len()); + for (index, target, input) in reads { let image = match target { ReadTarget::Account(target) => { self.client @@ -131,22 +142,24 @@ impl CellStorage { "committed read image is missing".into(), )); }; - Ok((index, image)) - }, - )) + group.push((index, image)); + } + Ok::<_, StorageError>(group) + })) .buffer_unordered(8) .collect::>() .await; let mut items = vec![None; count]; - for result in images { - let (index, image) = result?; - let slot = items - .get_mut(usize::from(index)) - .ok_or_else(|| StorageError::Internal("invalid transaction read index".into()))?; - if slot.replace(image).is_some() { - return Err(StorageError::Internal( - "duplicate transaction read index".into(), - )); + for group in images { + for (index, image) in group? { + let slot = items.get_mut(usize::from(index)).ok_or_else(|| { + StorageError::Internal("invalid transaction read index".into()) + })?; + if slot.replace(image).is_some() { + return Err(StorageError::Internal( + "duplicate transaction read index".into(), + )); + } } } for item in &items { @@ -181,15 +194,90 @@ impl CellStorage { "coordinator rejected read result acknowledgement".into(), )); } - if let Err(error) = self - .finish_decided_cross_cell_transaction( - account_id, - &identity.routing_key, - identity.transaction_id, + // The coordinator acknowledgement is durable. Release each participant + // directly from the targets already read above, avoiding a second + // coordinator status/participant-discovery round trip while preserving + // the same receipt ordering and restart recovery contract. + let cleanup = stream::iter(participant_targets.into_iter().map(|(position, target)| { + let coordinator = coordinator.clone(); + let identity = identity.clone(); + async move { + let participant_cell = match &target { + ReadTarget::Account(target) | ReadTarget::Data(target) => { + *target.cell_id().as_bytes() + } + }; + let coordinator_cell = *coordinator.cell_id().as_bytes(); + let mutation = mutation_identity()?; + let mut admission_retries = 0; + let released = loop { + let input = Json(ReadTransactionInput { + transaction_id: identity.transaction_id, + coordinator_cell, + }); + let result = match &target { + ReadTarget::Account(target) => { + self.client + .command::(target, mutation, input) + .await + } + ReadTarget::Data(target) => { + self.client + .command::( + target, mutation, input, + ) + .await + } + }; + match result { + Err(InvocationError::NotStarted(cellule_runtime::Error::Capacity(_))) + if admission_retries < 4 => + { + admission_retries += 1; + tokio::time::sleep(Duration::from_millis(2 << admission_retries)).await; + } + other => break other.map_err(cell_error)?, + } + }; + if !released.output.0 { + return Err(StorageError::Internal( + "participant rejected read result release".into(), + )); + } + Ok::<_, StorageError>(ReadResultRelease { + position, + participant_cell, + sequence: released.receipt.commit_sequence, + }) + } + })) + .buffer_unordered(8) + .collect::>() + .await; + let releases = cleanup.into_iter().collect::, _>>()?; + let recorded = self + .client + .command::( + &coordinator, + mutation_identity()?, + Json(RecordReadResultReleasesInput { + account_id: account_id.to_owned(), + transaction_id: identity.transaction_id, + routing_key: identity.routing_key, + releases, + }), ) .await - { - tracing::warn!(%error, "transaction read result cleanup remains pending"); + .map_err(cell_error)?; + if recorded.output.0.iter().any(|outcome| { + !matches!( + outcome, + CoordinatorPhaseOutcome::Recorded | CoordinatorPhaseOutcome::Replay + ) + }) { + return Err(StorageError::Internal( + "coordinator rejected read result release".into(), + )); } validate_read_size(items) } diff --git a/src/bin/beyonddb.rs b/src/bin/beyonddb.rs index 32b9437..fc03ff3 100644 --- a/src/bin/beyonddb.rs +++ b/src/bin/beyonddb.rs @@ -4,17 +4,21 @@ use std::{error::Error, io, io::Read, net::SocketAddr, path::PathBuf, sync::Arc, use beyonddb::{ APPLICATION_ID, Beyonddb, BeyonddbPeers, CellAuthorizationStore, CellCredentialStore, - CellInitialPartitionProvisioner, CellStorage, NodeLeasePublisher, build_http_state_with_cache, - measured_node_capacity, shutdown_serving_node, + CellInitialPartitionProvisioner, CellStorage, NodeLeasePublisher, PeerNodeLogTransport, + build_http_state_with_cache, measured_node_capacity, shutdown_serving_node, }; use cellule_app::CellApplication; -use cellule_host::{CellNode, CellNodeBuilder, CellNodeTaskGroup}; +use cellule_host::{CellNode, CellNodeBuilder, CellNodeTaskGroup, FOLLOWER_STORE_COMPONENT}; use cellule_peer_http::{LoadedPeerTls, PeerTlsIdentity}; use cellule_runtime::{ - SqlWorkerPool, + NodeLeaseGuard, SqlWorkerPool, + follower::FollowerStore, identity::{Digest, NodeId, SessionId}, - ltx::{CellStorageLayout, DiskBudget, Host}, - node::{NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain}, + ltx::{CellStorageLayout, DiskBudget, Host, Limits}, + node::{ + NODE_LOG_PROTOCOL_VERSION, NodeAdvertisement, NodeCapacity, NodeDirectory, + NodeFailureDomain, + }, registry::BuildDescriptor, }; use cellule_store::{Store, provider_store::build_url_object_store}; @@ -35,6 +39,16 @@ struct Config { node_id: Uuid, data_dir: PathBuf, disk_budget_bytes: u64, + /// Optional persistent follower-lane budget. Does not enable fleet proofs. + #[serde(default)] + follower_store_bytes: Option, + #[serde(default = "default_node_retained_bytes")] + node_retained_bytes: usize, + #[serde(default = "default_max_active_cells")] + max_active_cells: usize, + /// Optional SQL worker override. The runtime caps this at sixteen workers. + #[serde(default)] + sql_workers: Option, encryption_key_file: PathBuf, region: String, peer_bind: SocketAddr, @@ -55,8 +69,9 @@ struct Config { initial_partitions: u16, #[serde(default = "default_split_threshold")] split_threshold_bytes: u64, - /// Enable ExtendDB's stale-while-revalidate auth and table metadata caches. - /// Changes made on another node become visible after the cache TTL. + /// Enable stale-while-revalidate auth and table metadata caches plus the + /// complete-page routed table cache. Changes made on another node become + /// visible after the cache TTL or a stale-route rejection. #[serde(default)] auth_cache_enabled: bool, bootstrap: Option, @@ -80,6 +95,16 @@ const fn default_split_threshold() -> u64 { 256 * 1024 * 1024 } +const fn default_node_retained_bytes() -> usize { + 1024 * 1024 * 1024 +} + +const fn default_max_active_cells() -> usize { + 128 +} + +const MAX_SQL_WORKERS: usize = 16; + #[tokio::main] async fn main() -> ServerResult<()> { tracing_subscriber::fmt() @@ -115,8 +140,19 @@ async fn main() -> ServerResult<()> { } async fn serve(config: Config, bootstrap_secret: Option>) -> ServerResult<()> { - if config.disk_budget_bytes == 0 || config.split_threshold_bytes == 0 { - return Err(invalid("disk budget and split threshold must be positive").into()); + if config.disk_budget_bytes == 0 + || config.node_retained_bytes == 0 + || config.split_threshold_bytes == 0 + || config.max_active_cells == 0 + || config.follower_store_bytes == Some(0) + || config + .sql_workers + .is_some_and(|workers| !(1..=MAX_SQL_WORKERS).contains(&workers)) + { + return Err(invalid( + "disk, optional follower-store, retained-byte, split, and active-cell budgets must be positive; sql_workers must be between 1 and 16", + ) + .into()); } if !["s3://", "gs://", "az://"] .iter() @@ -186,13 +222,24 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S let session = SessionId::from_bytes(*session_uuid.as_bytes()); let session_dir = config.data_dir.join(session_uuid.to_string()); tokio::fs::create_dir_all(&config.data_dir).await?; - let node = CellNodeBuilder::new(Arc::clone(&application)) - .with_runtime(SqlWorkerPool::new(4, 64)?, 256 * 1024 * 1024) + let sql_workers = match config.sql_workers { + Some(workers) => SqlWorkerPool::new(workers, config.max_active_cells)?, + None => SqlWorkerPool::for_system(config.max_active_cells)?, + }; + let mut builder = CellNodeBuilder::new(Arc::clone(&application)) + .with_runtime(sql_workers, config.node_retained_bytes) .with_replica_host( Host::default().with_local_disk_budget(DiskBudget::new(config.disk_budget_bytes)), ) - .with_session(session) - .build()?; + .with_session(session); + if let Some(bytes) = config.follower_store_bytes { + builder = builder.with_follower_store( + config.data_dir.join("follower-store"), + Limits::default(), + DiskBudget::new(bytes), + ); + } + let node = builder.build()?; let node_shutdown = CancellationToken::new(); let tasks = node.install_task_group(CancellationToken::new(), node_shutdown.clone())?; let node_id = NodeId::from_bytes(*config.node_id.as_bytes()); @@ -202,13 +249,22 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S let fleet = tls.fleet(); let capacity_runtime = node.runtime(); let capacity_dir = config.data_dir.clone(); + let recovery_capable = config.follower_store_bytes.is_some(); let modules = application.registry().module_digests(); let published = NodeLeasePublisher::new(directory.clone(), move |now, expires| { let (capacity, placement) = if capacity_runtime.is_shutting_down() { (NodeCapacity::default(), None) } else { match measured_node_capacity(&capacity_dir, capacity_runtime.stats()) { - Ok((capacity, placement)) => (capacity, Some(placement)), + Ok((mut capacity, placement)) => { + // A persistent follower store permits fenced recovery claims. + // Zero advertised follower bytes still prevents recruitment + // until follower-backed serving has a tested recovery path. + if recovery_capable { + capacity.log_protocol = NODE_LOG_PROTOCOL_VERSION; + } + (capacity, Some(placement)) + } Err(error) => { // Observation failure disables placement, not the serving lease. // Never renew a stale sample with the next advertisement's time. @@ -241,7 +297,8 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S }) .publish() .await?; - node.install_node_lease_for_startup(published.guard())?; + let follower_guard = published.guard(); + node.install_node_lease_for_startup(follower_guard.clone())?; // Publication during drain still needs the node lease. The host cancels // lease maintenance only after the runtime and its durable log close. tasks.spawn_lease_maintenance(async move { published.run(&node_shutdown).await })?; @@ -256,6 +313,7 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S directory.clone(), application, session, + follower_guard, tls, peer_listener, public_listener, @@ -291,6 +349,7 @@ async fn serve_ready( directory: NodeDirectory, application: Arc, session: SessionId, + follower_guard: NodeLeaseGuard, tls: LoadedPeerTls, peer_listener: TcpListener, public_listener: TcpListener, @@ -298,25 +357,37 @@ async fn serve_ready( encryption_key: [u8; 32], bootstrap_secret: Option>, ) -> ServerResult<()> { - let peers = Arc::new(BeyonddbPeers::new( - node, + let mut peers = BeyonddbPeers::new(node, layout.clone(), directory.clone(), session, &tls)?; + let mut recovery_transport = None; + if let Some(store) = node.owned_component::(FOLLOWER_STORE_COMPONENT) { + let node_id = NodeId::from_bytes(*config.node_id.as_bytes()); + recovery_transport = Some(Arc::new( + PeerNodeLogTransport::new( + directory.clone(), + tls.client_identity(), + session, + node_id, + follower_guard.clone(), + ) + .with_local_follower_store(store.clone()), + )); + peers = peers.with_follower_store(node_id, store, follower_guard); + } + let peers = Arc::new(peers); + let mut provisioner = CellInitialPartitionProvisioner::new( + node.runtime(), + application, layout.clone(), - directory.clone(), session, - &tls, - )?); - let provisioner = Arc::new( - CellInitialPartitionProvisioner::new( - node.runtime(), - application, - layout.clone(), - session, - config.peer_endpoint.clone(), - session_dir, - )? - .with_initial_partition_count(config.initial_partitions)? - .with_peers(peers.clone()), - ); + config.peer_endpoint.clone(), + session_dir, + )? + .with_initial_partition_count(config.initial_partitions)? + .with_peers(peers.clone()); + if let Some(transport) = recovery_transport { + provisioner = provisioner.with_node_log_recovery(transport)?; + } + let provisioner = Arc::new(provisioner); for account_id in &config.owned_accounts { provisioner .recover_configured_account(account_id, &directory) @@ -327,7 +398,7 @@ async fn serve_ready( .recover_owned_credential(key_id, &directory) .await?; } - let client = peers.client(provisioner.clone()); + let client = peers.client_with_cache(provisioner.clone(), config.auth_cache_enabled); if let (Some(bootstrap), Some(secret)) = (config.bootstrap.as_ref(), bootstrap_secret) { let policy = std::fs::read_to_string(&bootstrap.policy_file)?; CellCredentialStore::new(client.clone(), layout.clone(), encryption_key) diff --git a/src/item_storage.rs b/src/item_storage.rs index 9271a9c..0fed6ab 100644 --- a/src/item_storage.rs +++ b/src/item_storage.rs @@ -5,7 +5,7 @@ use serde::{Serialize, de::DeserializeOwned}; use crate::{Error, Result, SqlBatch, SqlResultSet, SqlValue, table::statement}; -const CHUNK_BYTES: usize = 256 * 1024; +pub(crate) const CHUNK_BYTES: usize = 256 * 1024; pub(crate) enum StoredValue<'a> { Account { @@ -98,6 +98,14 @@ impl StoredValue<'_> { ) -> Result<()> { let (table, predicate, mut parameters) = self.address(); let bytes = serde_json::to_vec(item)?; + if bytes.len() <= CHUNK_BYTES { + parameters[0] = SqlValue::Blob(bytes); + context.sql(&statement( + &format!("UPDATE {table} SET item = ?1 WHERE {predicate}"), + parameters, + ))?; + return Ok(()); + } parameters[0] = SqlValue::Integer( i64::try_from(bytes.len()).map_err(|_| Error::Command("item size overflow"))?, ); diff --git a/src/items.rs b/src/items.rs index 0dfd830..ff8a07e 100644 --- a/src/items.rs +++ b/src/items.rs @@ -47,56 +47,87 @@ impl Command for PutItem { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - let Some(table) = command_unrouted_table(context, &input.table_name)? else { - return Ok(CommandResult::Rejected(Json( - ItemMutationOutcome::TableNotFound, - ))); - }; - if table.id != input.table_id { - return Ok(CommandResult::Rejected(Json( - ItemMutationOutcome::TableNotFound, - ))); - } - if !valid_item(&input.item, &table) { - return Ok(CommandResult::Rejected(Json( - ItemMutationOutcome::InvalidItem, - ))); - } - let key = item_key(&input.item, &table.key_schema)?; - if transaction::key_locked(context, &table.id, &key)? { - return Ok(CommandResult::Rejected(Json(ItemMutationOutcome::Conflict))); - } - let old = command_item(context, &table.id, &key)?; - if let Some(condition) = input.condition { - let empty = Item::new(); - match condition.evaluate(old.as_ref().unwrap_or(&empty)) { - Ok(true) => {} - Ok(false) => { - return Ok(CommandResult::Rejected(Json( - ItemMutationOutcome::ConditionFailed(old), - ))); - } - Err(message) => { - return Ok(CommandResult::Rejected(Json( - ItemMutationOutcome::InvalidExpression(message), - ))); - } + execute_put_item(context, input, true) + } +} + +/// Replace an item without returning its previous image. +pub struct PutItemNoReturn; + +impl Command for PutItemNoReturn { + const MODULE: &'static str = MODULE; + const ID: u32 = 50; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + execute_put_item(context, input, false) + } +} + +fn execute_put_item( + context: &mut CommandContext<'_, '_>, + input: PutItemInput, + return_old: bool, +) -> Result>> { + let Some(table) = command_unrouted_table(context, &input.table_name)? else { + return Ok(CommandResult::Rejected(Json( + ItemMutationOutcome::TableNotFound, + ))); + }; + if table.id != input.table_id { + return Ok(CommandResult::Rejected(Json( + ItemMutationOutcome::TableNotFound, + ))); + } + if !valid_item(&input.item, &table) { + return Ok(CommandResult::Rejected(Json( + ItemMutationOutcome::InvalidItem, + ))); + } + let key = item_key(&input.item, &table.key_schema)?; + if transaction::key_locked(context, &table.id, &key)? { + return Ok(CommandResult::Rejected(Json(ItemMutationOutcome::Conflict))); + } + let needs_old = return_old || input.condition.is_some() || table.stream.is_some(); + let old = if needs_old { + command_item(context, &table.id, &key)? + } else { + None + }; + if let Some(condition) = input.condition { + let empty = Item::new(); + match condition.evaluate(old.as_ref().unwrap_or(&empty)) { + Ok(true) => {} + Ok(false) => { + return Ok(CommandResult::Rejected(Json( + ItemMutationOutcome::ConditionFailed(old), + ))); + } + Err(message) => { + return Ok(CommandResult::Rejected(Json( + ItemMutationOutcome::InvalidExpression(message), + ))); } } - write_item(context, &table, &key, &input.item)?; - crate::stream_journal::append( - context, - &table.id, - &table.key_schema, - table.stream.as_ref(), - old.as_ref(), - Some(&input.item), - 0, - )?; - Ok(CommandResult::Success(Json(ItemMutationOutcome::Applied( - old, - )))) } + write_item(context, &table, &key, &input.item)?; + crate::stream_journal::append( + context, + &table.id, + &table.key_schema, + table.stream.as_ref(), + old.as_ref(), + Some(&input.item), + 0, + )?; + Ok(CommandResult::Success(Json(ItemMutationOutcome::Applied( + if return_old { old } else { None }, + )))) } /// One item deletion in a named table. @@ -128,56 +159,88 @@ impl Command for DeleteItem { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - let Some(table) = command_unrouted_table(context, &input.table_name)? else { - return Ok(CommandResult::Rejected(Json( - ItemMutationOutcome::TableNotFound, - ))); - }; - if table.id != input.table_id { - return Ok(CommandResult::Rejected(Json( - ItemMutationOutcome::TableNotFound, - ))); - } - if !valid_key(&input.key, &table) { - return Ok(CommandResult::Rejected(Json( - ItemMutationOutcome::InvalidItem, - ))); - } - let key = item_key(&input.key, &table.key_schema)?; - if transaction::key_locked(context, &table.id, &key)? { - return Ok(CommandResult::Rejected(Json(ItemMutationOutcome::Conflict))); - } - let old = command_item(context, &table.id, &key)?; - if let Some(condition) = input.condition { - let empty = Item::new(); - match condition.evaluate(old.as_ref().unwrap_or(&empty)) { - Ok(true) => {} - Ok(false) => { - return Ok(CommandResult::Rejected(Json( - ItemMutationOutcome::ConditionFailed(old), - ))); - } - Err(message) => { - return Ok(CommandResult::Rejected(Json( - ItemMutationOutcome::InvalidExpression(message), - ))); - } + let return_old = input.return_old; + execute_delete_item(context, input, return_old) + } +} + +/// Delete an item without reserving an item-sized result envelope. +pub struct DeleteItemNoReturn; + +impl Command for DeleteItemNoReturn { + const MODULE: &'static str = MODULE; + const ID: u32 = 55; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + execute_delete_item(context, input, false) + } +} + +fn execute_delete_item( + context: &mut CommandContext<'_, '_>, + input: DeleteItemInput, + return_old: bool, +) -> Result>> { + let Some(table) = command_unrouted_table(context, &input.table_name)? else { + return Ok(CommandResult::Rejected(Json( + ItemMutationOutcome::TableNotFound, + ))); + }; + if table.id != input.table_id { + return Ok(CommandResult::Rejected(Json( + ItemMutationOutcome::TableNotFound, + ))); + } + if !valid_key(&input.key, &table) { + return Ok(CommandResult::Rejected(Json( + ItemMutationOutcome::InvalidItem, + ))); + } + let key = item_key(&input.key, &table.key_schema)?; + if transaction::key_locked(context, &table.id, &key)? { + return Ok(CommandResult::Rejected(Json(ItemMutationOutcome::Conflict))); + } + let needs_old = return_old || input.condition.is_some() || table.stream.is_some(); + let old = if needs_old { + command_item(context, &table.id, &key)? + } else { + None + }; + if let Some(condition) = input.condition { + let empty = Item::new(); + match condition.evaluate(old.as_ref().unwrap_or(&empty)) { + Ok(true) => {} + Ok(false) => { + return Ok(CommandResult::Rejected(Json( + ItemMutationOutcome::ConditionFailed(old), + ))); + } + Err(message) => { + return Ok(CommandResult::Rejected(Json( + ItemMutationOutcome::InvalidExpression(message), + ))); } } - delete_item(context, &table, &key)?; - crate::stream_journal::append( - context, - &table.id, - &table.key_schema, - table.stream.as_ref(), - old.as_ref(), - None, - 0, - )?; - Ok(CommandResult::Success(Json(ItemMutationOutcome::Applied( - if input.return_old { old } else { None }, - )))) } + delete_item(context, &table, &key)?; + crate::stream_journal::append( + context, + &table.id, + &table.key_schema, + table.stream.as_ref(), + old.as_ref(), + None, + 0, + )?; + Ok(CommandResult::Success(Json(ItemMutationOutcome::Applied( + if return_old { old } else { None }, + )))) } /// One atomic item update in a named table. @@ -200,6 +263,8 @@ pub enum UpdateItemOutcome { Conflict, /// The update committed with both item images. Applied { old: Option, new: Item }, + /// The update committed without returning either item image. + AppliedNoReturn, /// The table does not exist. TableNotFound, /// The key or resulting item violates the table contract. @@ -224,68 +289,99 @@ impl Command for UpdateItem { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - let Some(table) = command_unrouted_table(context, &input.table_name)? else { - return Ok(CommandResult::Rejected(Json( - UpdateItemOutcome::TableNotFound, - ))); - }; - if table.id != input.table_id { - return Ok(CommandResult::Rejected(Json( - UpdateItemOutcome::TableNotFound, - ))); - } - if !valid_key(&input.key, &table) { - return Ok(CommandResult::Rejected(Json( - UpdateItemOutcome::InvalidItem, - ))); - } - let key = item_key(&input.key, &table.key_schema)?; - if transaction::key_locked(context, &table.id, &key)? { - return Ok(CommandResult::Rejected(Json(UpdateItemOutcome::Conflict))); - } - let old = command_item(context, &table.id, &key)?; - if let Some(condition) = input.condition { - let empty = Item::new(); - match condition.evaluate(old.as_ref().unwrap_or(&empty)) { - Ok(true) => {} - Ok(false) => { - return Ok(CommandResult::Rejected(Json( - UpdateItemOutcome::ConditionFailed(old), - ))); - } - Err(message) => { - return Ok(CommandResult::Rejected(Json( - UpdateItemOutcome::InvalidExpression(message), - ))); - } + execute_update_item(context, input, true) + } +} + +/// Apply an update without returning either item image. +pub struct UpdateItemNoReturn; + +impl Command for UpdateItemNoReturn { + const MODULE: &'static str = MODULE; + const ID: u32 = 51; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + execute_update_item(context, input, false) + } +} + +fn execute_update_item( + context: &mut CommandContext<'_, '_>, + input: UpdateItemInput, + return_images: bool, +) -> Result>> { + let Some(table) = command_unrouted_table(context, &input.table_name)? else { + return Ok(CommandResult::Rejected(Json( + UpdateItemOutcome::TableNotFound, + ))); + }; + if table.id != input.table_id { + return Ok(CommandResult::Rejected(Json( + UpdateItemOutcome::TableNotFound, + ))); + } + if !valid_key(&input.key, &table) { + return Ok(CommandResult::Rejected(Json( + UpdateItemOutcome::InvalidItem, + ))); + } + let key = item_key(&input.key, &table.key_schema)?; + if transaction::key_locked(context, &table.id, &key)? { + return Ok(CommandResult::Rejected(Json(UpdateItemOutcome::Conflict))); + } + let mut old = command_item(context, &table.id, &key)?; + if let Some(condition) = input.condition { + let empty = Item::new(); + match condition.evaluate(old.as_ref().unwrap_or(&empty)) { + Ok(true) => {} + Ok(false) => { + return Ok(CommandResult::Rejected(Json( + UpdateItemOutcome::ConditionFailed(old), + ))); + } + Err(message) => { + return Ok(CommandResult::Rejected(Json( + UpdateItemOutcome::InvalidExpression(message), + ))); } } - let mut new = old.clone().unwrap_or_else(|| input.key.clone()); - if let Err(message) = input.update.apply(&mut new, &table.attribute_definitions) { - return Ok(CommandResult::Rejected(Json( - UpdateItemOutcome::InvalidExpression(message), - ))); - } - if !valid_item(&new, &table) || item_key(&new, &table.key_schema)? != key { - return Ok(CommandResult::Rejected(Json( - UpdateItemOutcome::InvalidItem, - ))); - } - write_item(context, &table, &key, &new)?; - crate::stream_journal::append( - context, - &table.id, - &table.key_schema, - table.stream.as_ref(), - old.as_ref(), - Some(&new), - 0, - )?; - Ok(CommandResult::Success(Json(UpdateItemOutcome::Applied { - old, - new, - }))) } + let mut new = if return_images || table.stream.is_some() { + old.clone().unwrap_or_else(|| input.key.clone()) + } else { + old.take().unwrap_or_else(|| input.key.clone()) + }; + if let Err(message) = input.update.apply(&mut new, &table.attribute_definitions) { + return Ok(CommandResult::Rejected(Json( + UpdateItemOutcome::InvalidExpression(message), + ))); + } + if !valid_item(&new, &table) || item_key(&new, &table.key_schema)? != key { + return Ok(CommandResult::Rejected(Json( + UpdateItemOutcome::InvalidItem, + ))); + } + write_item(context, &table, &key, &new)?; + crate::stream_journal::append( + context, + &table.id, + &table.key_schema, + table.stream.as_ref(), + old.as_ref(), + Some(&new), + 0, + )?; + Ok(CommandResult::Success(Json(if return_images { + UpdateItemOutcome::Applied { old, new } + } else { + UpdateItemOutcome::AppliedNoReturn + }))) } /// One keyed item read in a named table. @@ -456,10 +552,33 @@ impl Command for TransactWrite { } } +/// Account-local transactional writes that do not return condition-failure images. +pub struct TransactWriteNoReturn; + +impl Command for TransactWriteNoReturn { + const MODULE: &'static str = MODULE; + const ID: u32 = 53; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + let outcome = transaction::write_without_old_images(context, input.operations)?; + if outcome != TransactionOutcome::Applied { + return Ok(CommandResult::Rejected(Json(outcome))); + } + Ok(CommandResult::Success(Json(TransactionOutcome::Applied))) + } +} + pub(crate) mod transaction; pub use transaction::{ PrepareAccountTransaction, PrepareAccountTransactionInput, ReadAccountTransaction, ReadAccountTransactionResult, ReleaseAccountTransactionReads, ResolveAccountTransaction, + TransactRead, TransactReadQuery, TransactionReadOutcome, }; mod scan; @@ -500,17 +619,22 @@ fn write_item( let old = command_item(context, table_id, key)?; crate::global_index::outbox::enqueue(context, table, key, 0, old, Some(item.clone()))?; } + let bytes = serde_json::to_vec(item)?; + let inline = bytes.len() <= crate::item_storage::CHUNK_BYTES; context.sql(&statement( - "INSERT INTO ddb_items (table_id, item_key, partition_key, sort_key, item, logical_bytes) VALUES (?1, ?2, ?3, ?4, X'', ?5) ON CONFLICT(table_id, item_key) DO UPDATE SET partition_key = excluded.partition_key, sort_key = excluded.sort_key, item = excluded.item, logical_bytes = excluded.logical_bytes", + "INSERT INTO ddb_items (table_id, item_key, partition_key, sort_key, item, logical_bytes) VALUES (?1, ?2, ?3, ?4, ?5, ?6) ON CONFLICT(table_id, item_key) DO UPDATE SET partition_key = excluded.partition_key, sort_key = excluded.sort_key, item = excluded.item, logical_bytes = excluded.logical_bytes", vec![ SqlValue::Text(table_id.into()), SqlValue::Blob(key.to_vec()), SqlValue::Blob(crate::partition::key::partition_key_bytes(item, &table.key_schema)?), SqlValue::Blob(crate::partition::key::index_key(item, &table.key_schema)?.1), + SqlValue::Blob(if inline { bytes } else { Vec::new() }), SqlValue::Integer(crate::statistics::item_bytes(item)?), ], ))?; - crate::item_storage::StoredValue::Account { table_id, key }.write(context, item)?; + if !inline { + crate::item_storage::StoredValue::Account { table_id, key }.write(context, item)?; + } crate::secondary_index::write(context, table, key, item) } diff --git a/src/items/transaction.rs b/src/items/transaction.rs index 1d607b8..aad60de 100644 --- a/src/items/transaction.rs +++ b/src/items/transaction.rs @@ -185,6 +185,146 @@ pub(super) fn write( Ok(TransactionOutcome::Applied) } +pub(super) fn write_without_old_images( + context: &mut CommandContext<'_, '_>, + operations: Vec, +) -> Result { + Ok(match write(context, operations)? { + TransactionOutcome::Applied => TransactionOutcome::Applied, + TransactionOutcome::Rejected { index, reason } => TransactionOutcome::Rejected { + index, + reason: without_old_image(reason), + }, + }) +} + +pub(super) fn without_old_image(reason: TransactionFailure) -> TransactionFailure { + match reason { + TransactionFailure::ConditionFailed(_) => TransactionFailure::ConditionFailed(None), + reason => reason, + } +} + +/// Result of an atomic read batch confined to one account Cell. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub enum TransactionReadOutcome { + /// Every requested image was read from one Cell snapshot. + Applied(Vec>), + /// No image was returned because one operation failed validation or locking. + Rejected { + /// Position of the failing read. + index: usize, + /// Validation or transaction conflict reason. + reason: TransactionFailure, + }, +} + +/// Read a transaction batch atomically when every item belongs to one account Cell. +pub struct TransactRead; + +impl Command for TransactRead { + const MODULE: &'static str = MODULE; + const ID: u32 = 52; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + let staged = match stage(context, input.operations)? { + Ok(staged) => staged, + Err((index, reason)) => { + return Ok(CommandResult::Rejected(Json( + TransactionReadOutcome::Rejected { index, reason }, + ))); + } + }; + let images = staged.into_iter().map(|image| image.image).collect(); + Ok(CommandResult::Success(Json( + TransactionReadOutcome::Applied(images), + ))) + } +} + +/// Read a transaction batch through Cellule's read-only query path. +/// +/// The query callback runs on the Cell worker, so all item and lock reads are +/// serialized with commands while avoiding a durable mutation publication. +pub struct TransactReadQuery; + +impl Query for TransactReadQuery { + const MODULE: &'static str = MODULE; + const ID: u32 = 54; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute(context: &mut QueryContext<'_>, Json(input): Self::Input) -> Result { + let images = match query_stage(context, input.operations)? { + Ok(images) => images, + Err((index, reason)) => { + return Ok(Json(TransactionReadOutcome::Rejected { index, reason })); + } + }; + Ok(Json(TransactionReadOutcome::Applied(images))) + } +} + +type ReadStage = std::result::Result>, (usize, TransactionFailure)>; + +fn query_stage( + context: &mut QueryContext<'_>, + operations: Vec, +) -> Result { + if operations.is_empty() || operations.len() > 100 { + return Ok(Err(( + 0, + TransactionFailure::Validation("transaction operation count is outside 1..=100".into()), + ))); + } + let mut images = Vec::with_capacity(operations.len()); + let mut read_bytes = 0; + for (index, operation) in operations.into_iter().enumerate() { + let TransactionOperation::Read(input) = operation else { + return Ok(Err(( + index, + TransactionFailure::Validation( + "read-only transaction contains a non-read operation".into(), + ), + ))); + }; + let invalid = |message: &str| (index, TransactionFailure::Validation(message.into())); + let Some(table) = query_unrouted_table(context, &input.table_name)? else { + return Ok(Err(invalid("table does not exist or has a data route"))); + }; + if table.id != input.table_id { + return Ok(Err(invalid("table identity is stale"))); + } + if !valid_key(&input.key, &table) { + return Ok(Err(invalid("item violates table schema"))); + } + let key = item_key(&input.key, &table.key_schema)?; + if read_key_conflict(context, &table.id, &key)?.is_some() { + return Ok(Err((index, TransactionFailure::Conflict))); + } + let image = crate::item_storage::StoredValue::Account { + table_id: &table.id, + key: &key, + } + .read(|batch| context.sql(batch))?; + read_bytes += image + .as_ref() + .map_or(0, extenddb_core::types::item_size_bytes); + if read_bytes > 4 * 1024 * 1024 { + return Ok(Err(invalid("transaction read exceeds 4 MiB"))); + } + images.push(image); + } + Ok(Ok(images)) +} + /// Prepare read or write operations on unrouted tables in one account Cell. #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] pub struct PrepareAccountTransactionInput { @@ -206,7 +346,7 @@ impl Command for PrepareAccountTransaction { const MODULE: &'static str = MODULE; const ID: u32 = 21; const CODEC_VERSION: u32 = 1; - type Input = Json; + type Input = Json>; type Output = Json; fn execute( diff --git a/src/lib.rs b/src/lib.rs index 089a60d..aea7921 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -40,8 +40,10 @@ pub use partition::*; pub use provision::*; pub use routing::*; pub use server::{ - BeyonddbPeerScope, BeyonddbPeers, NodeLeasePublisher, PublishedNodeLease, build_http_state, - build_http_state_with_cache, measured_node_capacity, shutdown_serving_node, + BeyonddbPeerScope, BeyonddbPeers, NodeLeasePublisher, PeerNodeDurabilityProvider, + PeerNodeLogTransport, PublishedNodeLease, PublishedNodeLogAuthority, build_http_state, + build_http_state_with_cache, measured_node_capacity, recover_fenced_node_log, + shutdown_serving_node, }; pub use split::*; pub use stream_journal::{ @@ -58,8 +60,8 @@ pub use table::*; pub use transaction_coordinator::*; pub use transaction_token::TransactionToken; pub use transaction_transport::{ - MultipartTransactionCommand, TransactionPayloadChunk, TransactionPayloadRef, - UploadTransactionPayload, + MultipartTransactionCommand, TransactionCommandInput, TransactionPayloadChunk, + TransactionPayloadRef, UploadTransactionPayload, }; pub use ttl::*; @@ -107,6 +109,15 @@ static SCHEMA: std::sync::LazyLock = std::sync::LazyLock::new(|| { ) }); const OPERATION_BYTES: u32 = 4 * 1024 * 1024 + 64 * 1024; +// Unconditional no-return mutations do not carry item images in their result. +// Keep their mailbox reservation small so concurrent writes are not limited by +// the generic 4 MiB command envelope. +const NO_RETURN_INPUT_BYTES: u32 = 1024 * 1024; +const NO_RETURN_OUTPUT_BYTES: u32 = 64 * 1024; +// A transaction write can return one failed item's old image, but never a +// successful item list. Keep the result envelope below the generic operation +// bound while retaining the full 4 MiB input budget for up to 100 operations. +const TRANSACTION_WRITE_OUTPUT_BYTES: u32 = 1024 * 1024; static NAMESPACES: [NamespaceDescriptor; 1] = [NamespaceDescriptor { id: NAMESPACE, @@ -128,11 +139,44 @@ const fn operation(id: u32) -> OperationDescriptor { } } -static COMMANDS: [OperationDescriptor; 28] = [ +const fn no_return_operation(id: u32) -> OperationDescriptor { + OperationDescriptor { + id, + codec_version: 1, + schema_min: 1, + schema_max: 1, + input_limit: NO_RETURN_INPUT_BYTES, + output_limit: NO_RETURN_OUTPUT_BYTES, + } +} + +const fn transaction_write_operation(id: u32) -> OperationDescriptor { + OperationDescriptor { + id, + codec_version: 1, + schema_min: 1, + schema_max: 1, + input_limit: OPERATION_BYTES, + output_limit: TRANSACTION_WRITE_OUTPUT_BYTES, + } +} + +const fn no_return_transaction_operation(id: u32) -> OperationDescriptor { + OperationDescriptor { + id, + codec_version: 1, + schema_min: 1, + schema_max: 1, + input_limit: OPERATION_BYTES, + output_limit: NO_RETURN_OUTPUT_BYTES, + } +} + +static COMMANDS: [OperationDescriptor; 33] = [ operation(1), operation(2), operation(3), - operation(5), + transaction_write_operation(5), operation(7), operation(8), operation(9), @@ -176,8 +220,13 @@ static COMMANDS: [OperationDescriptor; 28] = [ operation(43), operation(44), operation(45), + no_return_operation(50), + no_return_operation(51), + operation(52), + no_return_transaction_operation(53), + no_return_operation(55), ]; -static QUERIES: [OperationDescriptor; 31] = [ +static QUERIES: [OperationDescriptor; 32] = [ operation(4), operation(7), OperationDescriptor { @@ -227,6 +276,7 @@ static QUERIES: [OperationDescriptor; 31] = [ operation(47), operation(48), operation(49), + operation(54), ]; /// Statically linked account application. @@ -401,8 +451,11 @@ impl cellule_runtime::registry::CellModule for AccountModule { registry.bind_query::()?; registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::>()?; registry.bind_command::()?; registry.bind_command::()?; @@ -414,6 +467,9 @@ impl cellule_runtime::registry::CellModule for AccountModule { registry.bind_query::()?; registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_query::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; diff --git a/src/partition.rs b/src/partition.rs index 4d61f4e..5770ca7 100644 --- a/src/partition.rs +++ b/src/partition.rs @@ -53,7 +53,7 @@ static NAMESPACES: [NamespaceDescriptor; 1] = [NamespaceDescriptor { effect_targets: &[], dead_letter: None, }]; -static COMMANDS: [OperationDescriptor; 20] = [ +static COMMANDS: [OperationDescriptor; 25] = [ operation(1), operation(2), OperationDescriptor { @@ -65,7 +65,7 @@ static COMMANDS: [OperationDescriptor; 20] = [ operation(6), operation(7), operation(8), - operation(9), + crate::transaction_write_operation(9), operation(10), operation(11), operation(12), @@ -80,8 +80,16 @@ static COMMANDS: [OperationDescriptor; 20] = [ operation(18), operation(19), operation(20), + no_return_operation(21), + no_return_operation(22), + operation(23), + crate::no_return_transaction_operation(24), + OperationDescriptor { + codec_version: 2, + ..no_return_operation(52) + }, ]; -static QUERIES: [OperationDescriptor; 17] = [ +static QUERIES: [OperationDescriptor; 18] = [ operation(1), operation(2), operation(3), @@ -99,6 +107,7 @@ static QUERIES: [OperationDescriptor; 17] = [ operation(16), operation(17), operation(18), + operation(19), ]; const fn operation(id: u32) -> OperationDescriptor { @@ -112,6 +121,17 @@ const fn operation(id: u32) -> OperationDescriptor { } } +const fn no_return_operation(id: u32) -> OperationDescriptor { + OperationDescriptor { + id, + codec_version: 1, + schema_min: 1, + schema_max: 1, + input_limit: 1024 * 1024, + output_limit: 64 * 1024, + } +} + pub(crate) struct DataModule; impl cellule_runtime::registry::CellModule for DataModule { @@ -173,13 +193,19 @@ impl cellule_runtime::registry::CellModule for DataModule { registry.bind_query::()?; registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_query::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::>()?; @@ -941,76 +967,107 @@ impl Command for PartitionPut { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - let Some(spec) = indexes::command_spec(context)? else { - return Ok(CommandResult::Rejected(Json( - PartitionPutOutcome::NotInstalled, - ))); - }; - if spec.table.id != input.table_id || spec.epoch != input.epoch { - return Ok(CommandResult::Rejected(Json( - PartitionPutOutcome::StaleRoute, - ))); - } - match command_access(context)? { - AccessState::Serving => {} - AccessState::Sealed => { - return Ok(CommandResult::Rejected(Json(PartitionPutOutcome::Sealed))); - } - AccessState::Importing => { - return Ok(CommandResult::Rejected(Json(PartitionPutOutcome::NotReady))); - } - } - if !valid_item(&input.item, &spec.table) { - return Ok(CommandResult::Rejected(Json( - PartitionPutOutcome::InvalidItem, - ))); - } - let key = item_key(&input.item, &spec.table.key_schema)?; - if !spec.contains(data_key_hash( - &spec.table.id, - &input.item, - &spec.table.key_schema, - )?) { - return Ok(CommandResult::Rejected(Json( - PartitionPutOutcome::WrongPartition, - ))); + execute_partition_put(context, input, true) + } +} + +/// Replace one item without returning its previous image. +pub struct PartitionPutNoReturn; + +impl Command for PartitionPutNoReturn { + const MODULE: &'static str = DATA_MODULE; + const ID: u32 = 21; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + execute_partition_put(context, input, false) + } +} + +fn execute_partition_put( + context: &mut CommandContext<'_, '_>, + input: PartitionPutInput, + return_old: bool, +) -> Result>> { + let Some(spec) = indexes::command_spec(context)? else { + return Ok(CommandResult::Rejected(Json( + PartitionPutOutcome::NotInstalled, + ))); + }; + if spec.table.id != input.table_id || spec.epoch != input.epoch { + return Ok(CommandResult::Rejected(Json( + PartitionPutOutcome::StaleRoute, + ))); + } + match command_access(context)? { + AccessState::Serving => {} + AccessState::Sealed => { + return Ok(CommandResult::Rejected(Json(PartitionPutOutcome::Sealed))); } - if transaction::key_locked(context, &key)? { - return Ok(CommandResult::Rejected(Json( - PartitionPutOutcome::TransactionConflict, - ))); + AccessState::Importing => { + return Ok(CommandResult::Rejected(Json(PartitionPutOutcome::NotReady))); } - let old = command_item(context, &key)?; - if let Some(condition) = input.condition { - let empty = Item::new(); - match condition.evaluate(old.as_ref().unwrap_or(&empty)) { - Ok(true) => {} - Ok(false) => { - return Ok(CommandResult::Rejected(Json( - PartitionPutOutcome::ConditionFailed(old), - ))); - } - Err(message) => { - return Ok(CommandResult::Rejected(Json( - PartitionPutOutcome::InvalidExpression(message), - ))); - } + } + if !valid_item(&input.item, &spec.table) { + return Ok(CommandResult::Rejected(Json( + PartitionPutOutcome::InvalidItem, + ))); + } + let key = item_key(&input.item, &spec.table.key_schema)?; + if !spec.contains(data_key_hash( + &spec.table.id, + &input.item, + &spec.table.key_schema, + )?) { + return Ok(CommandResult::Rejected(Json( + PartitionPutOutcome::WrongPartition, + ))); + } + if transaction::key_locked(context, &key)? { + return Ok(CommandResult::Rejected(Json( + PartitionPutOutcome::TransactionConflict, + ))); + } + let needs_old = return_old || input.condition.is_some() || spec.table.stream.is_some(); + let old = if needs_old { + command_item(context, &key)? + } else { + None + }; + if let Some(condition) = input.condition { + let empty = Item::new(); + match condition.evaluate(old.as_ref().unwrap_or(&empty)) { + Ok(true) => {} + Ok(false) => { + return Ok(CommandResult::Rejected(Json( + PartitionPutOutcome::ConditionFailed(old), + ))); + } + Err(message) => { + return Ok(CommandResult::Rejected(Json( + PartitionPutOutcome::InvalidExpression(message), + ))); } } - write_item(context, key, &input.item, &spec.table, Some(spec.epoch))?; - crate::stream_journal::append( - context, - &spec.table.id, - &spec.table.key_schema, - spec.table.stream.as_ref(), - old.as_ref(), - Some(&input.item), - 0, - )?; - Ok(CommandResult::Success(Json(PartitionPutOutcome::Applied( - old, - )))) } + write_item(context, key, &input.item, &spec.table, Some(spec.epoch))?; + crate::stream_journal::append( + context, + &spec.table.id, + &spec.table.key_schema, + spec.table.stream.as_ref(), + old.as_ref(), + Some(&input.item), + 0, + )?; + Ok(CommandResult::Success(Json(PartitionPutOutcome::Applied( + if return_old { old } else { None }, + )))) } /// Delete one item under a verified data Cell epoch. @@ -1069,95 +1126,128 @@ impl Command for PartitionDelete { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - let Some(spec) = indexes::command_spec(context)? else { + let return_old = input.return_old; + execute_partition_delete(context, input, return_old) + } +} + +/// Delete a partition item without reserving an item-sized result envelope. +pub struct PartitionDeleteNoReturn; + +impl Command for PartitionDeleteNoReturn { + const MODULE: &'static str = DATA_MODULE; + const ID: u32 = 52; + const CODEC_VERSION: u32 = 2; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + execute_partition_delete(context, input, false) + } +} + +fn execute_partition_delete( + context: &mut CommandContext<'_, '_>, + input: PartitionDeleteInput, + return_old: bool, +) -> Result>> { + let Some(spec) = indexes::command_spec(context)? else { + return Ok(CommandResult::Rejected(Json( + PartitionDeleteOutcome::NotInstalled, + ))); + }; + if spec.table.id != input.table_id || spec.epoch != input.epoch { + return Ok(CommandResult::Rejected(Json( + PartitionDeleteOutcome::StaleRoute, + ))); + } + match command_access(context)? { + AccessState::Serving => {} + AccessState::Sealed => { return Ok(CommandResult::Rejected(Json( - PartitionDeleteOutcome::NotInstalled, + PartitionDeleteOutcome::Sealed, ))); - }; - if spec.table.id != input.table_id || spec.epoch != input.epoch { + } + AccessState::Importing => { return Ok(CommandResult::Rejected(Json( - PartitionDeleteOutcome::StaleRoute, + PartitionDeleteOutcome::NotReady, ))); } - match command_access(context)? { - AccessState::Serving => {} - AccessState::Sealed => { + } + if !valid_key(&input.key, &spec.table) { + return Ok(CommandResult::Rejected(Json( + PartitionDeleteOutcome::InvalidKey, + ))); + } + let key = item_key(&input.key, &spec.table.key_schema)?; + if !spec.contains(data_key_hash( + &spec.table.id, + &input.key, + &spec.table.key_schema, + )?) { + return Ok(CommandResult::Rejected(Json( + PartitionDeleteOutcome::WrongPartition, + ))); + } + if transaction::key_locked(context, &key)? { + return Ok(CommandResult::Rejected(Json( + PartitionDeleteOutcome::TransactionConflict, + ))); + } + let needs_old = + return_old || input.condition.is_some() || input.ttl || spec.table.stream.is_some(); + let old = if needs_old { + command_item(context, &key)? + } else { + None + }; + if let Some(condition) = input.condition { + let empty = Item::new(); + match condition.evaluate(old.as_ref().unwrap_or(&empty)) { + Ok(true) => {} + Ok(false) => { return Ok(CommandResult::Rejected(Json( - PartitionDeleteOutcome::Sealed, + PartitionDeleteOutcome::ConditionFailed(old), ))); } - AccessState::Importing => { + Err(message) => { return Ok(CommandResult::Rejected(Json( - PartitionDeleteOutcome::NotReady, + PartitionDeleteOutcome::InvalidExpression(message), ))); } } - if !valid_key(&input.key, &spec.table) { - return Ok(CommandResult::Rejected(Json( - PartitionDeleteOutcome::InvalidKey, - ))); - } - let key = item_key(&input.key, &spec.table.key_schema)?; - if !spec.contains(data_key_hash( + } + if input.ttl && !ttl::expired_for_configured_ttl(context, old.as_ref())? { + return Ok(CommandResult::Rejected(Json( + PartitionDeleteOutcome::ConditionFailed(old), + ))); + } + delete_item(context, &spec.table, &key, spec.epoch)?; + if input.ttl { + crate::stream_journal::append_ttl_delete( + context, &spec.table.id, - &input.key, &spec.table.key_schema, - )?) { - return Ok(CommandResult::Rejected(Json( - PartitionDeleteOutcome::WrongPartition, - ))); - } - if transaction::key_locked(context, &key)? { - return Ok(CommandResult::Rejected(Json( - PartitionDeleteOutcome::TransactionConflict, - ))); - } - let old = command_item(context, &key)?; - if let Some(condition) = input.condition { - let empty = Item::new(); - match condition.evaluate(old.as_ref().unwrap_or(&empty)) { - Ok(true) => {} - Ok(false) => { - return Ok(CommandResult::Rejected(Json( - PartitionDeleteOutcome::ConditionFailed(old), - ))); - } - Err(message) => { - return Ok(CommandResult::Rejected(Json( - PartitionDeleteOutcome::InvalidExpression(message), - ))); - } - } - } - if input.ttl && !ttl::expired_for_configured_ttl(context, old.as_ref())? { - return Ok(CommandResult::Rejected(Json( - PartitionDeleteOutcome::ConditionFailed(old), - ))); - } - delete_item(context, &spec.table, &key, spec.epoch)?; - if input.ttl { - crate::stream_journal::append_ttl_delete( - context, - &spec.table.id, - &spec.table.key_schema, - spec.table.stream.as_ref(), - old.as_ref(), - )?; - } else { - crate::stream_journal::append( - context, - &spec.table.id, - &spec.table.key_schema, - spec.table.stream.as_ref(), - old.as_ref(), - None, - 0, - )?; - } - Ok(CommandResult::Success(Json( - PartitionDeleteOutcome::Applied(if input.return_old { old } else { None }), - ))) + spec.table.stream.as_ref(), + old.as_ref(), + )?; + } else { + crate::stream_journal::append( + context, + &spec.table.id, + &spec.table.key_schema, + spec.table.stream.as_ref(), + old.as_ref(), + None, + 0, + )?; } + Ok(CommandResult::Success(Json( + PartitionDeleteOutcome::Applied(if return_old { old } else { None }), + ))) } /// Update one item under a verified data Cell epoch. @@ -1198,6 +1288,8 @@ impl PartitionUpdateInput { pub enum PartitionUpdateOutcome { /// The update committed with both item images. Applied { old: Option, new: Item }, + /// The update committed without returning either item image. + AppliedNoReturn, /// No partition contract is installed. NotInstalled, /// The table ID or data Cell epoch is stale. @@ -1232,94 +1324,126 @@ impl Command for PartitionUpdate { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - let Some(spec) = indexes::command_spec(context)? else { + execute_partition_update(context, input, true) + } +} + +/// Apply one update without returning either item image. +pub struct PartitionUpdateNoReturn; + +impl Command for PartitionUpdateNoReturn { + const MODULE: &'static str = DATA_MODULE; + const ID: u32 = 22; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + execute_partition_update(context, input, false) + } +} + +fn execute_partition_update( + context: &mut CommandContext<'_, '_>, + input: PartitionUpdateInput, + return_images: bool, +) -> Result>> { + let Some(spec) = indexes::command_spec(context)? else { + return Ok(CommandResult::Rejected(Json( + PartitionUpdateOutcome::NotInstalled, + ))); + }; + if spec.table.id != input.table_id || spec.epoch != input.epoch { + return Ok(CommandResult::Rejected(Json( + PartitionUpdateOutcome::StaleRoute, + ))); + } + match command_access(context)? { + AccessState::Serving => {} + AccessState::Sealed => { return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::NotInstalled, + PartitionUpdateOutcome::Sealed, ))); - }; - if spec.table.id != input.table_id || spec.epoch != input.epoch { + } + AccessState::Importing => { return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::StaleRoute, + PartitionUpdateOutcome::NotReady, ))); } - match command_access(context)? { - AccessState::Serving => {} - AccessState::Sealed => { + } + if !valid_key(&input.key, &spec.table) { + return Ok(CommandResult::Rejected(Json( + PartitionUpdateOutcome::InvalidItem, + ))); + } + let key = item_key(&input.key, &spec.table.key_schema)?; + if !spec.contains(data_key_hash( + &spec.table.id, + &input.key, + &spec.table.key_schema, + )?) { + return Ok(CommandResult::Rejected(Json( + PartitionUpdateOutcome::WrongPartition, + ))); + } + if transaction::key_locked(context, &key)? { + return Ok(CommandResult::Rejected(Json( + PartitionUpdateOutcome::TransactionConflict, + ))); + } + let mut old = command_item(context, &key)?; + if let Some(condition) = input.condition { + let empty = Item::new(); + match condition.evaluate(old.as_ref().unwrap_or(&empty)) { + Ok(true) => {} + Ok(false) => { return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::Sealed, + PartitionUpdateOutcome::ConditionFailed(old), ))); } - AccessState::Importing => { + Err(message) => { return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::NotReady, + PartitionUpdateOutcome::InvalidExpression(message), ))); } } - if !valid_key(&input.key, &spec.table) { - return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::InvalidItem, - ))); - } - let key = item_key(&input.key, &spec.table.key_schema)?; - if !spec.contains(data_key_hash( - &spec.table.id, - &input.key, - &spec.table.key_schema, - )?) { - return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::WrongPartition, - ))); - } - if transaction::key_locked(context, &key)? { - return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::TransactionConflict, - ))); - } - let old = command_item(context, &key)?; - if let Some(condition) = input.condition { - let empty = Item::new(); - match condition.evaluate(old.as_ref().unwrap_or(&empty)) { - Ok(true) => {} - Ok(false) => { - return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::ConditionFailed(old), - ))); - } - Err(message) => { - return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::InvalidExpression(message), - ))); - } - } - } - let mut new = old.clone().unwrap_or_else(|| input.key.clone()); - if let Err(message) = input - .update - .apply(&mut new, &spec.table.attribute_definitions) - { - return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::InvalidExpression(message), - ))); - } - if !valid_item(&new, &spec.table) || item_key(&new, &spec.table.key_schema)? != key { - return Ok(CommandResult::Rejected(Json( - PartitionUpdateOutcome::InvalidItem, - ))); - } - write_item(context, key, &new, &spec.table, Some(spec.epoch))?; - crate::stream_journal::append( - context, - &spec.table.id, - &spec.table.key_schema, - spec.table.stream.as_ref(), - old.as_ref(), - Some(&new), - 0, - )?; - Ok(CommandResult::Success(Json( - PartitionUpdateOutcome::Applied { old, new }, - ))) } + let mut new = if return_images || spec.table.stream.is_some() { + old.clone().unwrap_or_else(|| input.key.clone()) + } else { + old.take().unwrap_or_else(|| input.key.clone()) + }; + if let Err(message) = input + .update + .apply(&mut new, &spec.table.attribute_definitions) + { + return Ok(CommandResult::Rejected(Json( + PartitionUpdateOutcome::InvalidExpression(message), + ))); + } + if !valid_item(&new, &spec.table) || item_key(&new, &spec.table.key_schema)? != key { + return Ok(CommandResult::Rejected(Json( + PartitionUpdateOutcome::InvalidItem, + ))); + } + write_item(context, key, &new, &spec.table, Some(spec.epoch))?; + crate::stream_journal::append( + context, + &spec.table.id, + &spec.table.key_schema, + spec.table.stream.as_ref(), + old.as_ref(), + Some(&new), + 0, + )?; + Ok(CommandResult::Success(Json(if return_images { + PartitionUpdateOutcome::Applied { old, new } + } else { + PartitionUpdateOutcome::AppliedNoReturn + }))) } /// Read one item from a routed partition. @@ -1491,12 +1615,14 @@ fn write_item( let old = command_item(context, &key)?; crate::global_index::outbox::enqueue(context, table, &key, epoch, old, Some(item.clone()))?; } + let bytes = serde_json::to_vec(item)?; + let inline = bytes.len() <= crate::item_storage::CHUNK_BYTES; let (partition_key, sort_key) = index_key(item, &table.key_schema)?; let (ttl_generation, ttl_epoch) = ttl::write_values(context, item)?; context.sql(&statement( "INSERT INTO ddb_partition_items \ (item_key, partition_key, sort_key, item, ttl_generation, ttl_epoch, logical_bytes) \ - VALUES (?1, ?2, ?3, X'', ?4, ?5, ?6) ON CONFLICT(item_key) DO UPDATE SET \ + VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7) ON CONFLICT(item_key) DO UPDATE SET \ partition_key = excluded.partition_key, sort_key = excluded.sort_key, \ item = excluded.item, ttl_generation = excluded.ttl_generation, \ ttl_epoch = excluded.ttl_epoch, logical_bytes = excluded.logical_bytes", @@ -1504,12 +1630,15 @@ fn write_item( SqlValue::Blob(key.clone()), SqlValue::Blob(partition_key), SqlValue::Blob(sort_key), + SqlValue::Blob(if inline { bytes } else { Vec::new() }), ttl_generation, ttl_epoch, SqlValue::Integer(crate::statistics::item_bytes(item)?), ], ))?; - crate::item_storage::StoredValue::Partition(&key).write(context, item)?; + if !inline { + crate::item_storage::StoredValue::Partition(&key).write(context, item)?; + } crate::secondary_index::write(context, table, &key, item) } diff --git a/src/partition/transaction.rs b/src/partition/transaction.rs index 8aa4d25..512dade 100644 --- a/src/partition/transaction.rs +++ b/src/partition/transaction.rs @@ -3,7 +3,7 @@ use crate::participant::StagedEffect; use std::collections::HashSet; -use cellule_runtime::registry::{Command, CommandContext, CommandResult, QueryContext}; +use cellule_runtime::registry::{Command, CommandContext, CommandResult, Query, QueryContext}; use extenddb_core::types::Item; use serde::{Deserialize, Serialize}; @@ -59,42 +59,251 @@ impl Command for PartitionTransactWrite { type Input = Json; type Output = Json; + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + execute_write(context, input, false) + } +} + +/// Partition-local transactional writes that do not return condition-failure images. +pub struct PartitionTransactWriteNoReturn; + +impl Command for PartitionTransactWriteNoReturn { + const MODULE: &'static str = DATA_MODULE; + const ID: u32 = 24; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + execute_write(context, input, true) + } +} + +fn execute_write( + context: &mut CommandContext<'_, '_>, + input: PartitionTransactWriteInput, + strip_old_images: bool, +) -> Result>> { + let Some(spec) = super::indexes::command_spec(context)? else { + return Ok(rejected(PartitionTransactWriteOutcome::NotInstalled)); + }; + if spec.table.id != input.table_id || spec.epoch != input.epoch { + return Ok(rejected(PartitionTransactWriteOutcome::StaleRoute)); + } + match command_access(context)? { + AccessState::Serving => {} + AccessState::Sealed => return Ok(rejected(PartitionTransactWriteOutcome::Sealed)), + AccessState::Importing => return Ok(rejected(PartitionTransactWriteOutcome::NotReady)), + } + if input.operations.is_empty() || input.operations.len() > 100 { + return Ok(validation( + 0, + "transaction operation count is outside 1..=100", + )); + } + + let staged = match stage_operations(context, &spec, input.operations)? { + Ok(staged) => staged, + Err(reason) => return Ok(rejected(reason.single_outcome(strip_old_images))), + }; + apply_staged(context, &spec.table, spec.epoch, staged)?; + Ok(CommandResult::Success(Json( + PartitionTransactWriteOutcome::Applied, + ))) +} + +/// Result of an atomic read batch confined to one data Cell. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub enum PartitionTransactReadOutcome { + /// Every requested image was read from one Cell snapshot. + Applied(Vec>), + /// No image was returned because one operation failed validation or locking. + Rejected { + /// Position of the failing read. + index: usize, + /// Validation or transaction conflict reason. + reason: TransactionFailure, + }, + /// The data Cell has no installed partition. + NotInstalled, + /// Table identity or routing epoch changed. + StaleRoute, + /// The source has been sealed for a split. + Sealed, + /// An import-only child is not yet serving. + NotReady, + /// A key is outside this partition's range. + WrongPartition, +} + +/// Read a transaction batch atomically when every item belongs to one data Cell. +pub struct PartitionTransactRead; + +impl Command for PartitionTransactRead { + const MODULE: &'static str = DATA_MODULE; + const ID: u32 = 23; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + fn execute( context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { let Some(spec) = super::indexes::command_spec(context)? else { - return Ok(rejected(PartitionTransactWriteOutcome::NotInstalled)); + return Ok(rejected_read(PartitionTransactReadOutcome::NotInstalled)); }; - if spec.table.id != input.table_id { - return Ok(rejected(PartitionTransactWriteOutcome::StaleRoute)); - } - if spec.epoch != input.epoch { - return Ok(rejected(PartitionTransactWriteOutcome::StaleRoute)); + if spec.table.id != input.table_id || spec.epoch != input.epoch { + return Ok(rejected_read(PartitionTransactReadOutcome::StaleRoute)); } match command_access(context)? { AccessState::Serving => {} - AccessState::Sealed => return Ok(rejected(PartitionTransactWriteOutcome::Sealed)), - AccessState::Importing => return Ok(rejected(PartitionTransactWriteOutcome::NotReady)), + AccessState::Sealed => { + return Ok(rejected_read(PartitionTransactReadOutcome::Sealed)); + } + AccessState::Importing => { + return Ok(rejected_read(PartitionTransactReadOutcome::NotReady)); + } } if input.operations.is_empty() || input.operations.len() > 100 { - return Ok(validation( - 0, - "transaction operation count is outside 1..=100", - )); + return Ok(rejected_read(PartitionTransactReadOutcome::Rejected { + index: 0, + reason: TransactionFailure::Validation( + "transaction operation count is outside 1..=100".into(), + ), + })); } - let staged = match stage_operations(context, &spec, input.operations)? { Ok(staged) => staged, - Err(reason) => return Ok(rejected(reason.single_outcome())), + Err(reason) => { + return Ok(rejected_read(match reason { + StageError::StaleRoute => PartitionTransactReadOutcome::StaleRoute, + StageError::WrongPartition => PartitionTransactReadOutcome::WrongPartition, + StageError::Rejected { index, reason } => { + PartitionTransactReadOutcome::Rejected { index, reason } + } + })); + } }; - apply_staged(context, &spec.table, spec.epoch, staged)?; + let images = staged.into_iter().map(|image| image.image).collect(); Ok(CommandResult::Success(Json( - PartitionTransactWriteOutcome::Applied, + PartitionTransactReadOutcome::Applied(images), ))) } } +/// Read a transaction batch through Cellule's read-only query path. +pub struct PartitionTransactReadQuery; + +impl Query for PartitionTransactReadQuery { + const MODULE: &'static str = DATA_MODULE; + const ID: u32 = 19; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute(context: &mut QueryContext<'_>, Json(input): Self::Input) -> Result { + let rows = context.sql(&statement( + "SELECT spec FROM ddb_partition WHERE singleton = 1", + vec![], + ))?; + let Some(spec) = super::decode_spec(&rows[0])? else { + return Ok(Json(PartitionTransactReadOutcome::NotInstalled)); + }; + if spec.table.id != input.table_id || spec.epoch != input.epoch { + return Ok(Json(PartitionTransactReadOutcome::StaleRoute)); + } + match super::query_access(context)? { + AccessState::Serving => {} + AccessState::Sealed => return Ok(Json(PartitionTransactReadOutcome::Sealed)), + AccessState::Importing => return Ok(Json(PartitionTransactReadOutcome::NotReady)), + } + if input.operations.is_empty() || input.operations.len() > 100 { + return Ok(Json(PartitionTransactReadOutcome::Rejected { + index: 0, + reason: TransactionFailure::Validation( + "transaction operation count is outside 1..=100".into(), + ), + })); + } + let images = match query_stage_operations(context, &spec, input.operations)? { + Ok(images) => images, + Err(error) => { + return Ok(Json(match error { + StageError::StaleRoute => PartitionTransactReadOutcome::StaleRoute, + StageError::WrongPartition => PartitionTransactReadOutcome::WrongPartition, + StageError::Rejected { index, reason } => { + PartitionTransactReadOutcome::Rejected { index, reason } + } + })); + } + }; + Ok(Json(PartitionTransactReadOutcome::Applied(images))) + } +} + +fn query_stage_operations( + context: &mut QueryContext<'_>, + spec: &PartitionSpec, + operations: Vec, +) -> Result>, StageError>> { + let mut images = Vec::with_capacity(operations.len()); + let mut read_bytes = 0; + for (index, operation) in operations.into_iter().enumerate() { + let TransactionOperation::Read(input) = operation else { + return Ok(Err(stage_validation( + index, + "read-only transaction contains a non-read operation", + ))); + }; + if input.table_id != spec.table.id { + return Ok(Err(StageError::StaleRoute)); + } + if !valid_key(&input.key, &spec.table) { + return Ok(Err(stage_validation(index, "item violates table schema"))); + } + let key = item_key(&input.key, &spec.table.key_schema)?; + if !spec.contains(data_key_hash( + &spec.table.id, + &input.key, + &spec.table.key_schema, + )?) { + return Ok(Err(StageError::WrongPartition)); + } + if read_key_conflict(context, &key)?.is_some() { + return Ok(Err(StageError::Rejected { + index, + reason: TransactionFailure::Conflict, + })); + } + let image = + crate::item_storage::StoredValue::Partition(&key).read(|batch| context.sql(batch))?; + read_bytes += image + .as_ref() + .map_or(0, extenddb_core::types::item_size_bytes); + if read_bytes > 4 * 1024 * 1024 { + return Ok(Err(stage_validation( + index, + "transaction read exceeds 4 MiB", + ))); + } + images.push(image); + } + Ok(Ok(images)) +} + +fn rejected_read( + outcome: PartitionTransactReadOutcome, +) -> CommandResult> { + CommandResult::Rejected(Json(outcome)) +} + fn rejected( outcome: PartitionTransactWriteOutcome, ) -> CommandResult> { @@ -128,13 +337,18 @@ enum StageError { } impl StageError { - fn single_outcome(self) -> PartitionTransactWriteOutcome { + fn single_outcome(self, strip_old_images: bool) -> PartitionTransactWriteOutcome { match self { Self::StaleRoute => PartitionTransactWriteOutcome::StaleRoute, Self::WrongPartition => PartitionTransactWriteOutcome::WrongPartition, - Self::Rejected { index, reason } => { - PartitionTransactWriteOutcome::Rejected { index, reason } - } + Self::Rejected { index, reason } => PartitionTransactWriteOutcome::Rejected { + index, + reason: if strip_old_images { + without_old_image(reason) + } else { + reason + }, + }, } } @@ -149,6 +363,13 @@ impl StageError { } } +fn without_old_image(reason: TransactionFailure) -> TransactionFailure { + match reason { + TransactionFailure::ConditionFailed(_) => TransactionFailure::ConditionFailed(None), + reason => reason, + } +} + fn stage_operations( context: &mut CommandContext<'_, '_>, spec: &PartitionSpec, diff --git a/src/partition/transaction/participant.rs b/src/partition/transaction/participant.rs index f1af806..14c59d1 100644 --- a/src/partition/transaction/participant.rs +++ b/src/partition/transaction/participant.rs @@ -44,7 +44,7 @@ impl Command for PreparePartitionTransaction { const MODULE: &'static str = DATA_MODULE; const ID: u32 = 12; const CODEC_VERSION: u32 = 1; - type Input = Json; + type Input = Json>; type Output = Json; fn execute( diff --git a/src/provision.rs b/src/provision.rs index d080b9d..05c58d9 100644 --- a/src/provision.rs +++ b/src/provision.rs @@ -23,7 +23,7 @@ use cellule_runtime::client::{CellClient, InvocationError}; use cellule_runtime::control::{ControlState, Owner, authority::CellAuthority}; use cellule_runtime::identity::{CellTarget, IncarnationId, SessionId}; use cellule_runtime::ltx::{CellStorageLayout, Limits}; -use cellule_runtime::node::NodeDirectory; +use cellule_runtime::node::{NodeDirectory, log_transport::NodeLogTransport}; use cellule_runtime::recovery::manifest::RecoveryManifestStore; use extenddb_storage::BoxedFuture; use extenddb_storage::error::StorageError; @@ -31,12 +31,16 @@ use extenddb_storage::error::StorageError; use crate::backend::{InitialPartitionProvisioner, cell_error, mutation_identity}; use crate::{ DATA_MODULE, DescribeTable, InstallPartition, InstallPartitionOutcome, Json, ListTables, - ListTablesInput, ListTablesOutcome, PartitionInstall, PartitionSpec, RegisterCoordinatorShard, - RegisterCoordinatorShardInput, RoutePageInput, RoutePageOutcome, SplitPlan, TableRecord, - account_target, coordinator_target, credential_target, data_target, initialize_account, - initialize_coordinator, initialize_credentials, initialize_partition, + ListTablesInput, ListTablesOutcome, PartitionInstall, PartitionSpec, PeerNodeLogTransport, + RegisterCoordinatorShard, RegisterCoordinatorShardInput, RoutePageInput, RoutePageOutcome, + SplitPlan, TableRecord, account_target, coordinator_target, credential_target, data_target, + initialize_account, initialize_coordinator, initialize_credentials, initialize_partition, + recover_fenced_node_log, }; +const MAX_NODE_LOG_RECOVERY_TENANTS: usize = 4_096; +const MAX_NODE_LOG_RECOVERY_CELLS: usize = 65_536; + /// Position in an account capacity sweep. #[derive(Clone, Debug, PartialEq, Eq)] pub struct CapacityCursor { @@ -60,6 +64,7 @@ pub struct CellInitialPartitionProvisioner { transaction_recovery: transactions::CoordinatorRecovery, admission: tokio::sync::Mutex<()>, peers: Option>, + recovery_transport: Option>, } impl CellInitialPartitionProvisioner { @@ -88,6 +93,7 @@ impl CellInitialPartitionProvisioner { transaction_recovery: Default::default(), admission: Default::default(), peers: None, + recovery_transport: None, }) } @@ -99,6 +105,31 @@ impl CellInitialPartitionProvisioner { self } + /// Recover an expired owner's active node log before taking over its Cells. + /// + /// The transport must be bound to this provisioner's live boot session. + pub fn with_node_log_recovery( + mut self, + transport: Arc, + ) -> Result { + if transport.session() != self.session { + return Err(StorageError::Internal( + "node-log recovery transport session differs".into(), + )); + } + self.recovery_transport = Some(transport); + Ok(self) + } + + #[cfg(test)] + pub(crate) fn with_test_node_log_recovery( + mut self, + transport: Arc, + ) -> Self { + self.recovery_transport = Some(transport); + self + } + /// Start each new table with a power-of-two number of independently owned ranges. /// /// Accepted counts are 1 through 256. Each table persists its selected count; @@ -607,32 +638,71 @@ impl CellInitialPartitionProvisioner { let _admission = self.admission.lock().await; self.reclaim_settled_capacity(target).await?; let authority = CellAuthority::new(self.layout.clone()); - let observed = authority + let mut observed = authority .load(target.cell_id()) .await .map_err(provision_error)? .ok_or_else(|| StorageError::Transient("Cell has no authority".into()))?; - let former = observed + let former_session = observed .value() .owner .as_ref() + .map(|owner| owner.session) .ok_or_else(|| StorageError::Transient("Cell has no serving owner".into()))?; - if former.session == self.session { + if former_session == self.session { return Err(StorageError::Transient( "Cell cannot be taken over from this owner state".into(), )); } let now_ms = lease_time_ms()?; + let mut recovery_scratch = self.directory.clone(); let takeover = match nodes - .takeover_proof(former.session, self.session, now_ms) + .takeover_proof(former_session, self.session, now_ms) .await .map_err(provision_error)? { Some(proof) => proof, - None => nodes - .claim_expired_for_takeover(former.session, self.session, now_ms) + None => match nodes + .claim_expired_for_takeover(former_session, self.session, now_ms) .await - .map_err(provision_error)?, + { + Ok(proof) => proof, + Err(CellError::PendingPublication) => { + let transport = self + .recovery_transport + .as_ref() + .ok_or(CellError::PendingPublication) + .map_err(provision_error)?; + let fenced = nodes + .claim_expired_for_recovery(former_session, self.session, lease_time_ms()?) + .await + .map_err(provision_error)?; + recovery_scratch = self.directory.join("node-log-recovery"); + let recovered = recover_fenced_node_log( + nodes, + &self.layout, + transport.clone(), + fenced, + Limits::default(), + recovery_scratch.clone(), + MAX_NODE_LOG_RECOVERY_TENANTS, + MAX_NODE_LOG_RECOVERY_CELLS, + ) + .await + .map_err(provision_error)?; + // Recovery pins the overlay by changing Cell authority. + // The takeover must see that new control and its scratch. + observed = authority + .load(target.cell_id()) + .await + .map_err(provision_error)? + .ok_or_else(|| { + StorageError::Transient("recovered Cell has no authority".into()) + })?; + recovered.takeover + } + Err(error) => return Err(provision_error(error)), + }, }; let replica = CellReplica::new( self.layout.clone(), @@ -671,7 +741,7 @@ impl CellInitialPartitionProvisioner { self.layout.clone(), self.replica_limits(target).map_err(provision_error)?, ) - .with_recovery_scratch(self.directory.clone()), + .with_recovery_scratch(recovery_scratch), destination, owner, ) @@ -816,11 +886,11 @@ impl CellInitialPartitionProvisioner { if control.root.is_some() && match control.state { ControlState::Idle => control.owner.is_none(), - ControlState::Recovering => control + ControlState::Recovering | ControlState::Serving => control .owner .as_ref() .is_some_and(|owner| owner.session == self.session), - _ => false, + ControlState::Tombstoned => false, } { return self.activate_published(target, proof, observed).await; diff --git a/src/provision/residency.rs b/src/provision/residency.rs index 6c05228..85406bc 100644 --- a/src/provision/residency.rs +++ b/src/provision/residency.rs @@ -229,11 +229,14 @@ impl CellInitialPartitionProvisioner { control.root.is_some() && match control.state { ControlState::Idle => control.owner.is_none(), - ControlState::Recovering => control + // A settled local actor can release residency while its + // published control still says Serving. Reopen that exact + // root under the same owner session on the next request. + ControlState::Recovering | ControlState::Serving => control .owner .as_ref() .is_some_and(|owner| owner.session == self.session), - _ => false, + ControlState::Tombstoned => false, } }; // A local owner can drain after the caller's authority read. Recheck diff --git a/src/provision/transactions.rs b/src/provision/transactions.rs index 19fcd14..257a6f0 100644 --- a/src/provision/transactions.rs +++ b/src/provision/transactions.rs @@ -440,13 +440,38 @@ impl CellInitialPartitionProvisioner { storage: &CellStorage, nodes: &NodeDirectory, ) -> Result<(), StorageError> { - let admission = self - .recover_registered_partitions(account_id, client, nodes) - .await; - let resolution = self - .recover_registered_coordinators(account_id, client, storage, nodes) + for attempt in 0..3 { + // A takeover can complete its authority transition before the + // previous actor's worker state has settled. Recheck the account + // owner and repeat the idempotent recovery scan on a transient + // startup failure, rather than exiting with half the scan done. + if attempt > 0 { + tokio::time::sleep(Duration::from_millis(250 * attempt)).await; + } + let result = async { + if attempt > 0 { + self.recover_configured_account(account_id, nodes).await?; + } + let admission = self + .recover_registered_partitions(account_id, client, nodes) + .await; + let resolution = self + .recover_registered_coordinators(account_id, client, storage, nodes) + .await; + admission.and(resolution) + } .await; - admission.and(resolution) + match result { + Ok(()) => return Ok(()), + Err(StorageError::Transient(error)) if attempt < 2 => { + tracing::warn!(account_id, attempt = attempt + 1, %error, "registered account recovery retrying"); + } + Err(error) => return Err(error), + } + } + Err(StorageError::Internal( + "registered account recovery exhausted attempts without a result".into(), + )) } /// Recover idle or expired registered coordinators for a configured account. diff --git a/src/server.rs b/src/server.rs index 4024cbd..2fbc4e6 100644 --- a/src/server.rs +++ b/src/server.rs @@ -2,11 +2,20 @@ mod capacity; mod node_lease; +mod node_log_authority; +mod node_log_provider; +mod node_log_receiver; +mod node_log_recovery; +mod node_log_sender; mod peer_receiver; mod placement; pub use capacity::measured_node_capacity; pub use node_lease::{NodeLeasePublisher, PublishedNodeLease}; +pub use node_log_authority::PublishedNodeLogAuthority; +pub use node_log_provider::PeerNodeDurabilityProvider; +pub use node_log_recovery::recover_fenced_node_log; +pub use node_log_sender::PeerNodeLogTransport; use std::sync::Arc; @@ -14,7 +23,8 @@ use cellule_host::CellNode; use cellule_peer_http::{LoadedPeerTls, PeerHttpRoundTrip, PeerTargetScope}; use cellule_runtime::client::CellClient; use cellule_runtime::control::authority::CellAuthority; -use cellule_runtime::identity::{CellTarget, SessionId}; +use cellule_runtime::follower::FollowerStore; +use cellule_runtime::identity::{CellTarget, NodeId, SessionId}; use cellule_runtime::ltx::CellStorageLayout; use cellule_runtime::node::NodeDirectory; use cellule_runtime::peer::PeerSigner; @@ -95,6 +105,7 @@ pub struct BeyonddbPeers { layout: CellStorageLayout, registry: Arc, placement: Arc, + follower: Option, } impl BeyonddbPeers { @@ -135,11 +146,45 @@ impl BeyonddbPeers { signer, round_trip, }), + follower: None, }) } + /// Add a persistent follower lane to the private mTLS listener. + /// + /// This does not advertise follower capacity or enable fleet durability. + #[must_use] + pub fn with_follower_store( + mut self, + node: NodeId, + store: Arc, + guard: cellule_runtime::NodeLeaseGuard, + ) -> Self { + self.follower = Some(node_log_receiver::FollowerEndpoint::new( + self.placement.directory.clone(), + self.runtime.clone(), + node, + store, + guard, + )); + self + } + /// Build a client that places idle ranges/directories and recovers expired owners. pub fn client(&self, provisioner: Arc) -> CellClient { + self.client_with_cache(provisioner, false) + } + + /// Build a client with the opt-in short-lived local owner cache. + /// + /// The cache keeps resident handles for 500 ms while the Cell handle still + /// fences drained owners. Authority is re-read after expiry, so ownership + /// changes remain bounded by the cache window. + pub fn client_with_cache( + &self, + provisioner: Arc, + handle_cache_enabled: bool, + ) -> CellClient { let principal = peer_receiver::peer_principal(self.placement.directory.fleet(), self.placement.session); CellClient::peer( @@ -149,14 +194,22 @@ impl BeyonddbPeers { self.placement.round_trip.clone(), ) .with_local_resolver(Arc::new( - peer_receiver::LocalResolver::serving(self, provisioner) - .with_placement(self.placement.clone()), + peer_receiver::LocalResolver::serving_with_cache( + self, + provisioner, + handle_cache_enabled, + ) + .with_placement(self.placement.clone()), )) } /// Build the authenticated peer route; mount only on this identity's mTLS listener. pub fn router(&self, provisioner: Arc) -> axum::Router { - peer_receiver::peer_router(self, provisioner) + let router = peer_receiver::peer_router(self, provisioner); + match &self.follower { + Some(follower) => router.merge(node_log_receiver::router(follower.clone())), + None => router, + } } pub(crate) fn directory(&self) -> &NodeDirectory { @@ -257,7 +310,8 @@ pub fn build_http_state_with_cache( let storage: Arc = Arc::new( CellStorage::new(client.clone(), region) .with_transaction_coordinators(provisioner.clone()) - .with_initial_partitions(provisioner), + .with_initial_partitions(provisioner) + .with_route_cache(cache_enabled), ); let credentials: Arc = Arc::new(CellCredentialStore::new( client.clone(), diff --git a/src/server/node_lease.rs b/src/server/node_lease.rs index 559eb81..cde9d92 100644 --- a/src/server/node_lease.rs +++ b/src/server/node_lease.rs @@ -7,6 +7,7 @@ use std::{ use cellule_runtime::node::{NodeAdvertisement, NodeDirectory, VersionedNodeAdvertisement}; use cellule_runtime::{Error, NodeLeaseGuard, Result}; +use tokio::sync::Mutex; use tokio_util::sync::CancellationToken; // Use the runtime's maximum advertisement lifetime for storage refresh headroom. @@ -50,7 +51,8 @@ impl NodeLeasePublisher { let guard = NodeLeaseGuard::new(unix_time_ms()?, observed.advertisement().expires_at_ms())?; Ok(PublishedNodeLease { publisher: self, - observed, + session: observed.advertisement().session(), + observed: Arc::new(Mutex::new(observed)), guard, fence_on_drop: true, }) @@ -73,7 +75,8 @@ impl NodeLeasePublisher { /// Retained lease task for a serving Cell node. pub struct PublishedNodeLease { publisher: NodeLeasePublisher, - observed: VersionedNodeAdvertisement, + session: cellule_runtime::identity::SessionId, + observed: Arc>, guard: NodeLeaseGuard, fence_on_drop: bool, } @@ -85,6 +88,20 @@ impl PublishedNodeLease { self.guard.clone() } + /// Bind node-log CAS operations to this exact published boot session. + /// + /// The adapter reloads the authoritative record after heartbeat races; + /// it cannot publish once this lease guard is fenced. + #[must_use] + pub fn log_authority(&self) -> super::PublishedNodeLogAuthority { + super::PublishedNodeLogAuthority::new( + self.publisher.directory.clone(), + self.session, + self.guard.clone(), + Arc::clone(&self.observed), + ) + } + /// Refresh the authoritative lease until cancellation or terminal failure. /// /// Serving hosts retain this task in their lease-maintenance phase until @@ -134,15 +151,16 @@ impl PublishedNodeLease { self.guard.check()?; let now_ms = unix_time_ms()?; let next = self.publisher.advertisement(now_ms).await?; + let mut observed = self.observed.lock().await; self.guard.check()?; - let observed = self + let renewed = self .publisher .directory - .refresh(&self.observed, next, now_ms) + .refresh(&observed, next, now_ms) .await?; self.guard - .renew(unix_time_ms()?, observed.advertisement().expires_at_ms())?; - self.observed = observed; + .renew(unix_time_ms()?, renewed.advertisement().expires_at_ms())?; + *observed = renewed; Ok(()) } } diff --git a/src/server/node_log_authority.rs b/src/server/node_log_authority.rs new file mode 100644 index 0000000..ec490f5 --- /dev/null +++ b/src/server/node_log_authority.rs @@ -0,0 +1,239 @@ +//! Authoritative node-log transitions for a published BeyondDB boot session. + +use std::sync::Arc; + +use futures_util::future::BoxFuture; +use tokio::sync::Mutex; + +use cellule_runtime::{ + Error, NodeLeaseGuard, Result, + identity::{NodeId, SessionId}, + node::{NodeDirectory, VersionedNodeAdvertisement}, + node::{ + durability::NodeLogAuthority, + log::NodeLogRotationBarrier, + log_state::{NodeLogPhase, NodeLogStatus}, + }, +}; + +use super::node_lease::unix_time_ms; + +const MAX_CAS_ATTEMPTS: usize = 4; + +/// Directory-backed authority for one leased node session. +/// +/// Each operation reloads the current version so a concurrent heartbeat does +/// not overwrite or permanently obstruct an enrolled log transition. The +/// directory still validates every transition and compares the exact ETag. +#[derive(Clone)] +pub struct PublishedNodeLogAuthority { + directory: NodeDirectory, + session: SessionId, + guard: NodeLeaseGuard, + observed: Arc>, +} + +impl PublishedNodeLogAuthority { + pub(super) const fn session(&self) -> SessionId { + self.session + } + + pub(super) fn new( + directory: NodeDirectory, + session: SessionId, + guard: NodeLeaseGuard, + observed: Arc>, + ) -> Self { + Self { + directory, + session, + guard, + observed, + } + } + + async fn load(&self, observed: &mut VersionedNodeAdvertisement) -> Result { + self.guard.check()?; + let now_ms = unix_time_ms()?; + let current = self + .directory + .load(self.session, now_ms) + .await? + .ok_or(Error::Fenced)?; + self.guard.check()?; + *observed = current; + Ok(now_ms) + } + + /// Enroll a complete follower set for the next log epoch, if available. + /// + /// A heartbeat CAS race is retried against the latest signed record. A + /// previously enrolled matching epoch returns its authoritative members. + pub async fn recruit( + &self, + log_epoch: u64, + required_follower_bytes: u64, + live_node_limit: usize, + ) -> Result>> { + let mut observed = self.observed.lock().await; + let mut last_error = None; + for _ in 0..MAX_CAS_ATTEMPTS { + let now_ms = self.load(&mut observed).await?; + if let Some(log) = observed.advertisement().log() { + if log.epoch() != log_epoch || log.phase() != NodeLogPhase::Open || log.active() { + return Err(Error::Node("node session has a different or active log")); + } + return Ok(Some(log.members().to_vec())); + } + match self + .directory + .try_recruit_log( + &observed, + log_epoch, + required_follower_bytes, + live_node_limit, + now_ms, + ) + .await + { + Ok(Some(enrolled)) => { + *observed = enrolled; + self.guard.check()?; + let log = observed + .advertisement() + .log() + .ok_or(Error::Node("enrolled node log is missing"))?; + return Ok(Some(log.members().to_vec())); + } + Ok(None) => return Ok(None), + Err(error) => last_error = Some(error), + } + } + Err(cas_error( + last_error, + "node-log enrollment CAS did not complete", + )) + } + + async fn activate_epoch(&self, log_epoch: u64) -> Result<()> { + let mut observed = self.observed.lock().await; + let mut last_error = None; + for _ in 0..MAX_CAS_ATTEMPTS { + let now_ms = self.load(&mut observed).await?; + let log = exact_open_log(&observed, log_epoch)?; + if log.active() { + return Ok(()); + } + match self.directory.activate_log(&observed, now_ms).await { + Ok(updated) => { + *observed = updated; + self.guard.check()?; + return Ok(()); + } + Err(error) => last_error = Some(error), + } + } + Err(cas_error( + last_error, + "node-log activation CAS did not complete", + )) + } + + async fn advance_epoch_coverage(&self, log_epoch: u64, tiered_through: u64) -> Result<()> { + let mut observed = self.observed.lock().await; + let mut last_error = None; + for _ in 0..MAX_CAS_ATTEMPTS { + let now_ms = self.load(&mut observed).await?; + let log = exact_open_log(&observed, log_epoch)?; + if log.tiered_through() >= tiered_through { + return Ok(()); + } + match self + .directory + .advance_log_coverage(&observed, tiered_through, now_ms) + .await + { + Ok(updated) => { + *observed = updated; + self.guard.check()?; + return Ok(()); + } + Err(error) => last_error = Some(error), + } + } + Err(cas_error( + last_error, + "node-log coverage CAS did not complete", + )) + } + + async fn close_epoch(&self, barrier: &NodeLogRotationBarrier) -> Result<()> { + if barrier.leader_session() != self.session { + return Err(Error::Node( + "node-log close barrier belongs to another session", + )); + } + let mut observed = self.observed.lock().await; + let mut last_error = None; + let mut attempted = false; + for _ in 0..MAX_CAS_ATTEMPTS { + let now_ms = self.load(&mut observed).await?; + let Some(log) = observed.advertisement().log() else { + return if attempted { + Ok(()) + } else { + Err(Error::Node("node session has no enrolled log")) + }; + }; + if log.epoch() != barrier.log_epoch() { + return Err(Error::Node("node-log close barrier epoch differs")); + } + attempted = true; + match self.directory.close_log(&observed, barrier, now_ms).await { + Ok(updated) => { + *observed = updated; + self.guard.check()?; + return Ok(()); + } + Err(error) => last_error = Some(error), + } + } + Err(cas_error(last_error, "node-log close CAS did not complete")) + } +} + +fn cas_error(last_error: Option, message: &'static str) -> Error { + match last_error { + Some(error) => error, + None => Error::Node(message), + } +} + +fn exact_open_log(observed: &VersionedNodeAdvertisement, log_epoch: u64) -> Result<&NodeLogStatus> { + let log = observed + .advertisement() + .log() + .ok_or(Error::Node("node session has no enrolled log"))?; + if log.epoch() != log_epoch || log.phase() != NodeLogPhase::Open { + return Err(Error::Node("node-log epoch or phase differs")); + } + Ok(log) +} + +impl NodeLogAuthority for PublishedNodeLogAuthority { + fn activate<'a>(&'a self, log_epoch: u64) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { self.activate_epoch(log_epoch).await }) + } + + fn advance_coverage<'a>( + &'a self, + log_epoch: u64, + tiered_through: u64, + ) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { self.advance_epoch_coverage(log_epoch, tiered_through).await }) + } + + fn close<'a>(&'a self, barrier: &'a NodeLogRotationBarrier) -> BoxFuture<'a, Result<()>> { + Box::pin(async move { self.close_epoch(barrier).await }) + } +} diff --git a/src/server/node_log_provider.rs b/src/server/node_log_provider.rs new file mode 100644 index 0000000..2beba1e --- /dev/null +++ b/src/server/node_log_provider.rs @@ -0,0 +1,110 @@ +//! Product enrollment adapter for Cellule's host-owned node durability. + +use std::{ + future::Future, + pin::Pin, + sync::{ + Arc, + atomic::{AtomicU64, Ordering}, + }, +}; + +use cellule_host::{FacilityResult, NodeDurabilityProvider, NodeDurabilityRotation}; +use cellule_runtime::{ + Error, NodeLeaseGuard, Result, + fleet::telemetry::CellTelemetryHandle, + identity::{NodeId, SessionId}, + ltx::Limits, + node::{ + durability::{NodeDurabilityConfig, NodeLogAuthority}, + log_transport::NodeLogTransport, + }, +}; + +use super::{PeerNodeLogTransport, PublishedNodeLogAuthority}; + +/// Recruits authoritative follower sets for the host's durability supervisor. +/// +/// Constructing this adapter does not install it or enable follower proofs. +/// The serving binary must complete owner recovery before installing it. +pub struct PeerNodeDurabilityProvider { + authority: Arc, + transport: Arc, + session: SessionId, + node: NodeId, + guard: NodeLeaseGuard, + telemetry: CellTelemetryHandle, + next_epoch: AtomicU64, +} + +impl PeerNodeDurabilityProvider { + /// Binds recruitment to one published node boot and transport identity. + pub fn new( + authority: PublishedNodeLogAuthority, + transport: PeerNodeLogTransport, + session: SessionId, + node: NodeId, + guard: NodeLeaseGuard, + telemetry: CellTelemetryHandle, + ) -> Result { + if authority.session() != session + || transport.session() != session + || transport.node() != node + { + return Err(Error::PeerAuthorization( + "node-log provider identities differ", + )); + } + guard.check()?; + Ok(Self { + authority: Arc::new(authority), + transport: Arc::new(transport), + session, + node, + guard, + telemetry, + next_epoch: AtomicU64::new(1), + }) + } +} + +impl NodeDurabilityProvider for PeerNodeDurabilityProvider { + fn recruit( + self: Arc, + limits: Limits, + required_follower_bytes: u64, + live_node_limit: usize, + ) -> Pin>> + Send>> { + Box::pin(async move { + self.guard.check()?; + let epoch = self.next_epoch.load(Ordering::Acquire); + let Some(members) = self + .authority + .recruit(epoch, required_follower_bytes, live_node_limit) + .await? + else { + return Ok(None); + }; + self.guard.check()?; + let transport: Arc = self.transport.clone(); + let authority: Arc = self.authority.clone(); + Ok(Some(NodeDurabilityConfig::new( + self.session, + self.node, + epoch, + members, + transport, + authority, + self.guard.clone(), + limits, + self.telemetry.clone(), + )?)) + }) + } + + fn rotation_event(&self, event: NodeDurabilityRotation) { + if event == NodeDurabilityRotation::Started { + self.next_epoch.fetch_add(1, Ordering::AcqRel); + } + } +} diff --git a/src/server/node_log_receiver.rs b/src/server/node_log_receiver.rs new file mode 100644 index 0000000..70a7fbc --- /dev/null +++ b/src/server/node_log_receiver.rs @@ -0,0 +1,749 @@ +//! Bounded follower-lane requests on BeyondDB's mutual-TLS peer listener. + +use std::{sync::Arc, time::Duration}; + +use axum::{ + Router, + body::Bytes, + extract::{ConnectInfo, Request, State}, + http::{StatusCode, header}, + response::{IntoResponse, Response}, + routing::post, +}; +use cellule_peer_http::PeerTlsIdentity; +use cellule_runtime::{ + Error, NodeLeaseGuard, Result, + cell::actor::CellRuntime, + follower::{FollowerReceipt, FollowerStore, FollowerTailPage}, + identity::{Digest, NodeId, SessionId}, + node::NodeDirectory, +}; +use futures_util::StreamExt; +use tokio::sync::Semaphore; + +use super::node_lease::unix_time_ms; + +pub(super) const MEDIA_TYPE: &str = "application/vnd.beyonddb.node-log-v1"; +const MAGIC: &[u8; 4] = b"BNL1"; +pub(super) const MAX_REQUEST_BYTES: usize = 64 * 1024 * 1024 + 64 * 1024; +const MAX_APPEND_FRAMES: usize = 64; +const MAX_TAIL_FRAMES: usize = 4096; +const MAX_TAIL_BYTES: usize = 1024 * 1024; +pub(super) const MAX_RESPONSE_BYTES: usize = MAX_TAIL_BYTES + MAX_TAIL_FRAMES * 4 + 17; +const MAX_CONCURRENT_REQUESTS: usize = 8; +const REQUEST_TIMEOUT: Duration = Duration::from_secs(30); + +/// One local follower store and its current node-session fence. +#[derive(Clone)] +pub(super) struct FollowerEndpoint { + directory: NodeDirectory, + runtime: CellRuntime, + member: NodeId, + store: Arc, + guard: NodeLeaseGuard, + permits: Arc, +} + +impl FollowerEndpoint { + pub(super) fn new( + directory: NodeDirectory, + runtime: CellRuntime, + member: NodeId, + store: Arc, + guard: NodeLeaseGuard, + ) -> Self { + Self { + directory, + runtime, + member, + store, + guard, + permits: Arc::new(Semaphore::new(MAX_CONCURRENT_REQUESTS)), + } + } + + async fn dispatch(&self, request: WireRequest, identity: PeerTlsIdentity) -> Result> { + self.dispatch_with_identity(request, identity.certificate(), identity.public_key()) + .await + } + + async fn dispatch_with_identity( + &self, + request: WireRequest, + certificate: Digest, + public_key: [u8; 32], + ) -> Result> { + self.guard.check()?; + let now_ms = unix_time_ms()?; + self.directory + .peer_verifier(request.caller, certificate, public_key, now_ms) + .await?; + self.guard.check()?; + let now_ms = unix_time_ms()?; + let response = match request.operation { + Operation::Append(frames) => { + if request.caller != request.leader { + return Err(Error::PeerAuthorization( + "follower append caller is not leader", + )); + } + self.directory + .authorize_log_append( + request.leader, + self.member, + request.epoch, + request.argument, + now_ms, + ) + .await?; + self.guard.check()?; + encode_receipt( + self.store + .append(request.leader, request.epoch, frames, request.argument) + .await?, + ) + } + Operation::Seal => { + self.directory + .authorize_log_recovery( + request.leader, + request.caller, + self.member, + request.epoch, + now_ms, + ) + .await?; + self.guard.check()?; + encode_receipt(self.store.seal(request.leader, request.epoch).await?) + } + Operation::Retire => { + if request.caller != request.leader { + return Err(Error::PeerAuthorization( + "follower retire caller is not leader", + )); + } + self.directory + .authorize_log_retire( + request.leader, + self.member, + request.epoch, + request.argument, + now_ms, + ) + .await?; + self.guard.check()?; + encode_receipt( + self.store + .retire(request.leader, request.epoch, request.argument) + .await?, + ) + } + Operation::Tail => { + self.directory + .authorize_log_recovery( + request.leader, + request.caller, + self.member, + request.epoch, + now_ms, + ) + .await?; + self.guard.check()?; + encode_page( + self.store + .read_tail_page(request.leader, request.epoch, request.argument) + .await?, + )? + } + }; + self.guard.check()?; + Ok(response) + } +} + +pub(super) fn router(endpoint: FollowerEndpoint) -> Router { + Router::new() + .route("/internal/node-log/v1", post(receive)) + .with_state(endpoint) +} + +async fn receive( + State(endpoint): State, + ConnectInfo(identity): ConnectInfo, + request: Request, +) -> Response { + match tokio::time::timeout( + REQUEST_TIMEOUT, + receive_bounded(endpoint, identity, request), + ) + .await + { + Ok(Ok(encoded)) => ( + StatusCode::OK, + [(header::CONTENT_TYPE, MEDIA_TYPE)], + encoded, + ) + .into_response(), + Ok(Err(status)) => status.into_response(), + Err(_) => StatusCode::GATEWAY_TIMEOUT.into_response(), + } +} + +async fn receive_bounded( + endpoint: FollowerEndpoint, + identity: PeerTlsIdentity, + request: Request, +) -> std::result::Result, StatusCode> { + if request + .headers() + .get(header::CONTENT_TYPE) + .and_then(|value| value.to_str().ok()) + != Some(MEDIA_TYPE) + { + return Err(StatusCode::UNSUPPORTED_MEDIA_TYPE); + } + let wire_bytes = request + .headers() + .get(header::CONTENT_LENGTH) + .and_then(|value| value.to_str().ok()) + .and_then(|value| value.parse::().ok()) + .ok_or(StatusCode::LENGTH_REQUIRED)?; + if !(1..=MAX_REQUEST_BYTES).contains(&wire_bytes) { + return Err(StatusCode::PAYLOAD_TOO_LARGE); + } + let Ok(permit) = endpoint.permits.clone().try_acquire_owned() else { + return Err(StatusCode::SERVICE_UNAVAILABLE); + }; + let admission_bytes = wire_bytes.saturating_mul(2).saturating_add(64 * 1024); + let Ok(reservation) = endpoint.runtime.try_reserve_node_bytes(admission_bytes) else { + return Err(StatusCode::SERVICE_UNAVAILABLE); + }; + let mut encoded = Vec::with_capacity(wire_bytes.min(64 * 1024)); + let mut body = request.into_body().into_data_stream(); + while let Some(chunk) = body.next().await { + let chunk = chunk.map_err(|_| StatusCode::BAD_REQUEST)?; + let end = encoded + .len() + .checked_add(chunk.len()) + .ok_or(StatusCode::PAYLOAD_TOO_LARGE)?; + if end > wire_bytes { + return Err(StatusCode::PAYLOAD_TOO_LARGE); + } + encoded.extend_from_slice(&chunk); + } + if encoded.len() != wire_bytes { + return Err(StatusCode::BAD_REQUEST); + } + let request = + WireRequest::decode(Bytes::from(encoded)).map_err(|()| StatusCode::BAD_REQUEST)?; + let job = tokio::spawn(async move { + let _permit = permit; + let _reservation = reservation; + endpoint.dispatch(request, identity).await + }); + job.await + .map_err(|_| StatusCode::SERVICE_UNAVAILABLE)? + .map_err(|error| match error { + Error::PeerAuthorization(_) => StatusCode::FORBIDDEN, + Error::Node(_) | Error::Fenced => StatusCode::CONFLICT, + _ => StatusCode::SERVICE_UNAVAILABLE, + }) +} + +pub(super) struct WireRequest { + pub(super) caller: SessionId, + pub(super) leader: SessionId, + pub(super) epoch: u64, + pub(super) argument: u64, + pub(super) operation: Operation, +} + +pub(super) enum Operation { + Append(Vec), + Seal, + Retire, + Tail, +} + +impl WireRequest { + pub(super) fn encode(self) -> std::result::Result, ()> { + if self.epoch == 0 { + return Err(()); + } + let mut encoded = Vec::new(); + encoded.extend_from_slice(MAGIC); + let (tag, count) = match &self.operation { + Operation::Append(frames) if (1..=MAX_APPEND_FRAMES).contains(&frames.len()) => { + (1, frames.len()) + } + Operation::Seal if self.argument == 0 => (2, 0), + Operation::Retire => (3, 0), + Operation::Tail if self.argument != 0 => (4, 0), + _ => return Err(()), + }; + encoded.push(tag); + encoded.extend_from_slice(self.caller.as_bytes()); + encoded.extend_from_slice(self.leader.as_bytes()); + encoded.extend_from_slice(&self.epoch.to_be_bytes()); + encoded.extend_from_slice(&self.argument.to_be_bytes()); + encoded.extend_from_slice(&u32::try_from(count).map_err(|_| ())?.to_be_bytes()); + if let Operation::Append(frames) = self.operation { + for frame in frames { + if frame.is_empty() { + return Err(()); + } + encoded + .extend_from_slice(&u32::try_from(frame.len()).map_err(|_| ())?.to_be_bytes()); + encoded.extend_from_slice(&frame); + if encoded.len() > MAX_REQUEST_BYTES { + return Err(()); + } + } + } + Ok(encoded) + } + + fn decode(body: Bytes) -> std::result::Result { + let wire_bytes = body.len(); + if wire_bytes > MAX_REQUEST_BYTES { + return Err(()); + } + let mut reader = Reader::new(body); + if reader.take(4)?.as_ref() != MAGIC { + return Err(()); + } + let tag = reader.u8()?; + let caller = SessionId::from_bytes(reader.array_16()?); + let leader = SessionId::from_bytes(reader.array_16()?); + let epoch = reader.u64()?; + let argument = reader.u64()?; + let count = usize::try_from(reader.u32()?).map_err(|_| ())?; + if epoch == 0 { + return Err(()); + } + let operation = match tag { + 1 if (1..=MAX_APPEND_FRAMES).contains(&count) => { + let mut frames = Vec::with_capacity(count); + for _ in 0..count { + let length = usize::try_from(reader.u32()?).map_err(|_| ())?; + if length == 0 { + return Err(()); + } + frames.push(reader.take(length)?); + } + Operation::Append(frames) + } + 2 if count == 0 && argument == 0 => Operation::Seal, + 3 if count == 0 => Operation::Retire, + 4 if count == 0 && argument != 0 => Operation::Tail, + _ => return Err(()), + }; + reader.finish()?; + Ok(Self { + caller, + leader, + epoch, + argument, + operation, + }) + } +} + +struct Reader { + bytes: Bytes, + position: usize, +} + +impl Reader { + const fn new(bytes: Bytes) -> Self { + Self { bytes, position: 0 } + } + + fn take(&mut self, length: usize) -> std::result::Result { + let end = self.position.checked_add(length).ok_or(())?; + if end > self.bytes.len() { + return Err(()); + } + let value = self.bytes.slice(self.position..end); + self.position = end; + Ok(value) + } + + fn u8(&mut self) -> std::result::Result { + self.take(1)?.first().copied().ok_or(()) + } + + fn u32(&mut self) -> std::result::Result { + let bytes: [u8; 4] = self.take(4)?.as_ref().try_into().map_err(|_| ())?; + Ok(u32::from_be_bytes(bytes)) + } + + fn u64(&mut self) -> std::result::Result { + let bytes: [u8; 8] = self.take(8)?.as_ref().try_into().map_err(|_| ())?; + Ok(u64::from_be_bytes(bytes)) + } + + fn array_16(&mut self) -> std::result::Result<[u8; 16], ()> { + self.take(16)?.as_ref().try_into().map_err(|_| ()) + } + + fn finish(self) -> std::result::Result<(), ()> { + if self.position == self.bytes.len() { + Ok(()) + } else { + Err(()) + } + } +} + +fn encode_receipt(receipt: FollowerReceipt) -> Vec { + let mut encoded = Vec::with_capacity(21); + encoded.extend_from_slice(MAGIC); + encoded.push(1); + encoded.extend_from_slice(&receipt.base_sequence.to_be_bytes()); + encoded.extend_from_slice(&receipt.durable_through.to_be_bytes()); + encoded +} + +fn encode_page(page: FollowerTailPage) -> Result> { + if page.frames.len() > MAX_TAIL_FRAMES { + return Err(Error::Peer("follower tail page has too many frames")); + } + let bytes = page + .frames + .iter() + .try_fold(0_usize, |total, frame| total.checked_add(frame.len())); + let Some(bytes) = bytes.filter(|bytes| *bytes <= MAX_TAIL_BYTES) else { + return Err(Error::Peer("follower tail page exceeds byte limit")); + }; + let mut encoded = Vec::with_capacity(bytes.saturating_add(17 + page.frames.len() * 4)); + encoded.extend_from_slice(MAGIC); + encoded.push(2); + encoded.extend_from_slice( + &u32::try_from(page.frames.len()) + .map_err(|_| Error::Peer("follower tail frame count overflow"))? + .to_be_bytes(), + ); + encoded.extend_from_slice(&page.next_sequence.unwrap_or(0).to_be_bytes()); + for frame in page.frames { + encoded.extend_from_slice( + &u32::try_from(frame.len()) + .map_err(|_| Error::Peer("follower tail frame size overflow"))? + .to_be_bytes(), + ); + encoded.extend_from_slice(&frame); + } + Ok(encoded) +} + +pub(super) fn decode_receipt(body: Bytes) -> std::result::Result { + let mut reader = Reader::new(body); + if reader.take(4)?.as_ref() != MAGIC || reader.u8()? != 1 { + return Err(()); + } + let base_sequence = reader.u64()?; + let durable_through = reader.u64()?; + reader.finish()?; + if base_sequence == 0 || durable_through < base_sequence.saturating_sub(1) { + return Err(()); + } + Ok(FollowerReceipt { + base_sequence, + durable_through, + }) +} + +pub(super) fn decode_page(body: Bytes) -> std::result::Result { + if body.len() > MAX_RESPONSE_BYTES { + return Err(()); + } + let mut reader = Reader::new(body); + if reader.take(4)?.as_ref() != MAGIC || reader.u8()? != 2 { + return Err(()); + } + let count = usize::try_from(reader.u32()?).map_err(|_| ())?; + if count > MAX_TAIL_FRAMES { + return Err(()); + } + let next = reader.u64()?; + let mut bytes = 0_usize; + let mut frames = Vec::with_capacity(count); + for _ in 0..count { + let length = usize::try_from(reader.u32()?).map_err(|_| ())?; + bytes = bytes.checked_add(length).ok_or(())?; + if length == 0 || bytes > MAX_TAIL_BYTES { + return Err(()); + } + frames.push(reader.take(length)?); + } + reader.finish()?; + Ok(FollowerTailPage { + frames, + next_sequence: (next != 0).then_some(next), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use cellule_ltx::{Db, NodeFrameScope, encode_node_frame}; + use cellule_runtime::{ + SqlWorkerPool, + ltx::{CellStorageLayout, DiskBudget, Host, Limits}, + node::{NODE_LOG_PROTOCOL_VERSION, NodeAdvertisement, NodeCapacity, NodeFailureDomain}, + }; + use cellule_store::Store; + use ed25519_dalek::SigningKey; + use object_store::{memory::InMemory, path::Path}; + + fn request(tag: u8, count: u32, frames: &[&[u8]]) -> Bytes { + let mut encoded = Vec::new(); + encoded.extend_from_slice(MAGIC); + encoded.push(tag); + encoded.extend_from_slice(&[1; 16]); + encoded.extend_from_slice(&[2; 16]); + encoded.extend_from_slice(&3_u64.to_be_bytes()); + encoded.extend_from_slice(&1_u64.to_be_bytes()); + encoded.extend_from_slice(&count.to_be_bytes()); + for frame in frames { + encoded.extend_from_slice(&(frame.len() as u32).to_be_bytes()); + encoded.extend_from_slice(frame); + } + Bytes::from(encoded) + } + + #[test] + fn follower_wire_rejects_malformed_and_noncanonical_requests() { + let valid = WireRequest::decode(request(1, 1, &[b"frame"])).unwrap(); + assert!(matches!(valid.operation, Operation::Append(_))); + assert!(WireRequest::decode(request(1, 2, &[b"frame"])).is_err()); + assert!(WireRequest::decode(request(2, 0, &[])).is_err()); + assert!(WireRequest::decode(request(4, 0, &[])).is_ok()); + let mut trailing = request(1, 1, &[b"frame"]).to_vec(); + trailing.push(0); + assert!(WireRequest::decode(Bytes::from(trailing)).is_err()); + } + + #[test] + fn follower_wire_round_trips_transport_operations_and_responses() { + let caller = SessionId::from_bytes([1; 16]); + let leader = SessionId::from_bytes([2; 16]); + for operation in [ + Operation::Append(vec![Bytes::from_static(b"frame")]), + Operation::Seal, + Operation::Retire, + Operation::Tail, + ] { + let argument = if matches!(operation, Operation::Tail) { + 1 + } else { + 0 + }; + let request = WireRequest { + caller, + leader, + epoch: 3, + argument, + operation, + }; + let decoded = WireRequest::decode(Bytes::from(request.encode().unwrap())).unwrap(); + assert_eq!(decoded.caller, caller); + assert_eq!(decoded.leader, leader); + assert_eq!(decoded.epoch, 3); + assert_eq!(decoded.argument, argument); + } + let receipt = FollowerReceipt { + base_sequence: 2, + durable_through: 4, + }; + let decoded = decode_receipt(Bytes::from(encode_receipt(receipt))).unwrap(); + assert_eq!(decoded.base_sequence, 2); + assert_eq!(decoded.durable_through, 4); + let page = FollowerTailPage { + frames: vec![Bytes::from_static(b"frame")], + next_sequence: Some(5), + }; + let decoded = decode_page(Bytes::from(encode_page(page).unwrap())).unwrap(); + assert_eq!(decoded.frames, vec![Bytes::from_static(b"frame")]); + assert_eq!(decoded.next_sequence, Some(5)); + assert!(decode_receipt(Bytes::from_static(b"invalid")).is_err()); + assert!(decode_page(Bytes::from_static(b"invalid")).is_err()); + } + + fn advertisement( + node: NodeId, + session: SessionId, + key: u8, + follower: bool, + now_ms: i64, + ) -> NodeAdvertisement { + NodeAdvertisement::sign( + node, + session, + format!("https://node-{key}.internal:8081"), + Digest::from_bytes([80; 32]), + Digest::from_bytes([key; 32]), + Digest::from_bytes([81; 32]), + Digest::from_bytes([82; 32]), + &SigningKey::from_bytes(&[key; 32]), + 1, + now_ms, + now_ms + 15_000, + vec![Digest::from_bytes([86; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 16 << 20, + free_disk_bytes: 1 << 30, + follower_free_bytes: if follower { 1 << 30 } else { 0 }, + job_credits: 8, + log_protocol: if follower { + NODE_LOG_PROTOCOL_VERSION + } else { + 0 + }, + ..NodeCapacity::default() + }, + ) + .unwrap() + } + + #[tokio::test] + async fn authenticated_append_survives_reopen_and_rejects_wrong_peer() { + let limits = Limits::default(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("follower-receiver-test"), + [42; 16], + ); + let directory = NodeDirectory::new( + layout, + Digest::from_bytes([80; 32]), + Digest::from_bytes([81; 32]), + Digest::from_bytes([82; 32]), + ); + let leader = SessionId::from_bytes([1; 16]); + let follower = SessionId::from_bytes([2; 16]); + let member = NodeId::from_bytes([3; 16]); + let now_ms = unix_time_ms().unwrap(); + directory + .create(advertisement(member, follower, 22, true, now_ms), now_ms) + .await + .unwrap(); + let enrolled = directory + .create( + advertisement(NodeId::from_bytes([4; 16]), leader, 21, false, now_ms), + now_ms, + ) + .await + .unwrap(); + let enrolled = directory + .try_recruit_log(&enrolled, 2, 4096, 16, now_ms) + .await + .unwrap() + .unwrap(); + assert_eq!(enrolled.advertisement().log().unwrap().members(), &[member]); + + let source = tempfile::TempDir::new().unwrap(); + let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); + database + .transaction(|transaction| { + transaction.execute_batch( + "CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ + INSERT INTO events(body) VALUES ('one')", + ) + }) + .unwrap(); + let capture = database.capture().unwrap(); + let segment = capture.segments.first().unwrap(); + let frame = encode_node_frame( + NodeFrameScope { + leader_session: *leader.as_bytes(), + log_epoch: 2, + node_sequence: 1, + application: [3; 16], + cell: [4; 32], + incarnation: [5; 16], + cell_epoch: 6, + commit_sequence: 1, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + limits, + ) + .unwrap() + .encoded() + .clone(); + + let root = tempfile::TempDir::new().unwrap(); + let store = Arc::new( + FollowerStore::open(root.path().to_owned(), limits, DiskBudget::new(1 << 30)).unwrap(), + ); + let runtime = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(1, 8).unwrap(), + 64 << 20, + follower, + Host::default(), + ) + .unwrap(); + let guard = NodeLeaseGuard::new(now_ms, now_ms + 15_000).unwrap(); + let endpoint = FollowerEndpoint::new(directory, runtime, member, store.clone(), guard); + let append = || WireRequest { + caller: leader, + leader, + epoch: 2, + argument: 0, + operation: Operation::Append(vec![frame.clone()]), + }; + let leader_key = SigningKey::from_bytes(&[21; 32]).verifying_key().to_bytes(); + assert!( + endpoint + .dispatch_with_identity(append(), Digest::from_bytes([99; 32]), leader_key) + .await + .is_err() + ); + assert_eq!(store.retained_bytes(), 0); + for _ in 0..2 { + let result = endpoint + .dispatch_with_identity(append(), Digest::from_bytes([21; 32]), leader_key) + .await + .unwrap(); + assert_eq!(&result[result.len() - 8..], &1_u64.to_be_bytes()); + } + assert!( + endpoint + .dispatch_with_identity( + WireRequest { + epoch: 3, + ..append() + }, + Digest::from_bytes([21; 32]), + leader_key, + ) + .await + .is_err() + ); + assert!( + endpoint + .dispatch_with_identity( + WireRequest { + caller: follower, + leader, + epoch: 2, + argument: 0, + operation: Operation::Seal, + }, + Digest::from_bytes([22; 32]), + SigningKey::from_bytes(&[22; 32]).verifying_key().to_bytes(), + ) + .await + .is_err() + ); + drop(endpoint); + drop(store); + let reopened = + FollowerStore::open(root.path().to_owned(), limits, DiskBudget::new(1 << 30)).unwrap(); + assert_eq!(reopened.seal(leader, 2).await.unwrap().durable_through, 1); + assert_eq!(reopened.read_tail(leader, 2, 1).await.unwrap(), vec![frame]); + } +} diff --git a/src/server/node_log_recovery.rs b/src/server/node_log_recovery.rs new file mode 100644 index 0000000..89fc037 --- /dev/null +++ b/src/server/node_log_recovery.rs @@ -0,0 +1,721 @@ +//! Product-side inventory and overlay attachment for a fenced node log. + +use std::{ + collections::{BTreeMap, BTreeSet}, + path::PathBuf, + sync::Arc, +}; + +use cellule_ltx::NodeFrameScope; +use cellule_runtime::{ + Error, Result, + cell::catalog::CellCatalog, + control::authority::CellAuthority, + identity::{CellId, TenantId}, + ltx::{CellStorageLayout, Limits}, + node::{ + FencedNodeSession, NodeDirectory, + log_recovery::{ + CompletedNodeRecovery, NodeLogRecovery, RecoveryCell, RecoveryCoordinator, + recoverable_cells_from_scopes, + }, + log_transport::NodeLogTransport, + }, + recovery::manifest::RecoveryManifestStore, +}; +use futures_util::StreamExt; + +use super::node_lease::unix_time_ms; + +/// Seal a claimed dead owner's log, pin every untiered Cell overlay, then seal +/// the directory record so ordinary Cell takeover can proceed. +/// +/// The caller must retain its live node lease and provide a transport bound to +/// the claiming session. A failed inventory or overlay attachment leaves the +/// log unsealed; retry with a fresh fenced claim rather than taking over a Cell. +pub async fn recover_fenced_node_log( + directory: &NodeDirectory, + layout: &CellStorageLayout, + transport: Arc, + fenced: FencedNodeSession, + limits: Limits, + scratch: PathBuf, + max_tenants: usize, + max_cells: usize, +) -> Result { + if max_tenants == 0 || max_cells == 0 { + return Err(Error::Capacity("node-log recovery inventory bound is zero")); + } + tokio::fs::create_dir_all(&scratch) + .await + .map_err(|source| Error::Facility { + name: "node-log recovery scratch", + source: Box::new(source), + })?; + let recovery = NodeLogRecovery::from_fenced(transport, &fenced, limits)? + .with_recovery_scratch(scratch.clone()); + let sealed = recovery.ensure_sealed_bounded().await?; + let scopes = sealed.scopes(limits)?; + let cells = recovery_cells(layout, fenced.session(), &scopes, max_tenants, max_cells).await?; + let manifests = + RecoveryManifestStore::new(layout.clone(), limits).with_recovery_scratch(scratch); + let coordinator = RecoveryCoordinator::new(recovery, manifests); + let controls = coordinator + .recover_sealed(fenced.clone(), cells, sealed) + .await?; + coordinator + .finish(directory, fenced, controls, unix_time_ms()?) + .await +} + +async fn recovery_cells( + layout: &CellStorageLayout, + owner: cellule_runtime::identity::SessionId, + scopes: &[NodeFrameScope], + max_tenants: usize, + max_cells: usize, +) -> Result> { + let grouped = locate_scope_tenants(layout, scopes, max_tenants, max_cells).await?; + let authority = CellAuthority::new(layout.clone()); + let mut cells = Vec::with_capacity(scopes.len()); + for (tenant, tenant_scopes) in grouped { + let catalog = CellCatalog::new(layout.clone(), TenantId::from_bytes(tenant)); + let mut found = + recoverable_cells_from_scopes(&catalog, &authority, owner, &tenant_scopes, max_cells) + .await?; + cells.append(&mut found); + } + if cells.len() != scopes.len() { + return Err(Error::Catalog( + "node-log recovery Cell inventory is incomplete", + )); + } + Ok(cells) +} + +async fn locate_scope_tenants( + layout: &CellStorageLayout, + scopes: &[NodeFrameScope], + max_tenants: usize, + max_cells: usize, +) -> Result>> { + if max_tenants == 0 || max_cells == 0 || scopes.len() > max_cells { + return Err(Error::Capacity( + "node-log recovery inventory exceeds its limit", + )); + } + let mut unique = BTreeSet::new(); + let mut required_shards = BTreeSet::new(); + for scope in scopes { + if scope.application != *layout.application_id() { + return Err(Error::Catalog("node-log recovery application differs")); + } + if !unique.insert(scope.cell) { + return Err(Error::Catalog("node-log recovery Cell scope repeats")); + } + required_shards.insert(scope.cell[0]); + } + if scopes.is_empty() { + return Ok(BTreeMap::new()); + } + + let prefix = layout.catalog_tenants_prefix(); + let prefix_with_separator = format!("{prefix}/"); + let mut stream = layout.store().list_stream(&prefix); + let mut tenants = BTreeSet::new(); + let mut heads = BTreeMap::>::new(); + let max_heads = max_tenants + .checked_mul(256) + .ok_or(Error::Capacity("node-log recovery catalog head bound"))?; + let mut head_count = 0_usize; + while let Some(meta) = stream.next().await { + let meta = meta?; + head_count = head_count + .checked_add(1) + .ok_or(Error::Capacity("node-log recovery catalog head count"))?; + if head_count > max_heads { + return Err(Error::Capacity("node-log recovery catalog head limit")); + } + let (tenant, shard) = parse_catalog_head(&prefix_with_separator, meta.location.as_ref())?; + tenants.insert(tenant); + if tenants.len() > max_tenants { + return Err(Error::Capacity("node-log recovery tenant limit")); + } + if required_shards.contains(&shard) { + heads.entry(shard).or_default().insert(tenant); + } + } + + let mut grouped = BTreeMap::<[u8; 16], Vec>::new(); + for scope in scopes { + let mut found = None; + if let Some(candidates) = heads.get(&scope.cell[0]) { + for tenant in candidates { + let catalog = CellCatalog::new(layout.clone(), TenantId::from_bytes(*tenant)); + if catalog + .lookup(CellId::from_bytes(scope.cell)) + .await? + .is_some() + && found.replace(*tenant).is_some() + { + return Err(Error::Catalog( + "node-log recovery Cell has multiple tenants", + )); + } + } + } + let tenant = found.ok_or(Error::Catalog( + "node-log recovery Cell has no tenant catalog entry", + ))?; + grouped.entry(tenant).or_default().push(*scope); + } + Ok(grouped) +} + +fn parse_catalog_head(prefix: &str, path: &str) -> Result<([u8; 16], u8)> { + let relative = path + .strip_prefix(prefix) + .ok_or(Error::Catalog("node-log recovery catalog path differs"))?; + let mut parts = relative.split('/'); + let tenant = parts.next().ok_or(Error::Catalog( + "node-log recovery tenant path is incomplete", + ))?; + let shard = parts + .next() + .ok_or(Error::Catalog("node-log recovery shard path is incomplete"))?; + if parts.next() != Some("head.json") || parts.next().is_some() { + return Err(Error::Catalog( + "node-log recovery catalog head path differs", + )); + } + let tenant = decode_lower_hex::<16>(tenant)?; + let shard = decode_lower_hex::<1>(shard)?[0]; + Ok((tenant, shard)) +} + +fn decode_lower_hex(value: &str) -> Result<[u8; N]> { + if value.len() != N * 2 { + return Err(Error::Catalog("node-log recovery catalog hex length")); + } + let mut bytes = [0_u8; N]; + for (index, pair) in value.as_bytes().chunks_exact(2).enumerate() { + let high = nibble(pair[0]).ok_or(Error::Catalog("node-log recovery catalog hex"))?; + let low = nibble(pair[1]).ok_or(Error::Catalog("node-log recovery catalog hex"))?; + bytes[index] = (high << 4) | low; + } + Ok(bytes) +} + +const fn nibble(value: u8) -> Option { + match value { + b'0'..=b'9' => Some(value - b'0'), + b'a'..=b'f' => Some(value - b'a' + 10), + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use axum::body::Bytes; + use cellule_app::CellApplication; + use cellule_host::CellNodeBuilder; + use cellule_ltx::{CellReplica, encode_node_frame}; + use cellule_runtime::{ + SqlWorkerPool, + cell::catalog::{CatalogEntry, CatalogRole}, + control::Owner, + follower::FollowerReceipt, + identity::{CellTarget, Digest, IncarnationId, NodeId, SessionId, TenantId}, + ltx::{DiskBudget, Host}, + node::{ + NODE_LOG_PROTOCOL_VERSION, NodeAdvertisement, NodeCapacity, NodeFailureDomain, + log_transport::{ + AppendRequest, LocalFollowerTransport, RetireRequest, SealRequest, TailRequest, + }, + }, + registry::BuildDescriptor, + }; + use cellule_store::Store; + use ed25519_dalek::SigningKey; + use futures_util::future::BoxFuture; + use object_store::{memory::InMemory, path::Path as ObjectPath}; + use std::time::Duration; + + use crate::{ + APPLICATION, APPLICATION_ID, Beyonddb, CellInitialPartitionProvisioner, NAMESPACE, + account_target, initialize_account, + }; + + fn scope(cell: CellId) -> NodeFrameScope { + NodeFrameScope { + leader_session: [1; 16], + log_epoch: 2, + node_sequence: 1, + application: *APPLICATION_ID.as_bytes(), + cell: *cell.as_bytes(), + incarnation: [2; 16], + cell_epoch: 3, + commit_sequence: 4, + } + } + + #[tokio::test] + async fn scopes_resolve_only_their_durable_tenant_catalogs() { + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("node-log-recovery-inventory-test"), + *APPLICATION_ID.as_bytes(), + ); + let one = CellTarget::new( + TenantId::from_bytes([1; 16]), + APPLICATION, + NAMESPACE, + b"one", + ) + .unwrap(); + let two = CellTarget::new( + TenantId::from_bytes([2; 16]), + APPLICATION, + NAMESPACE, + b"two", + ) + .unwrap(); + for target in [&one, &two] { + CellCatalog::new(layout.clone(), target.tenant()) + .provision( + CatalogEntry::new(target, CatalogRole::Sql, Digest::from_bytes([3; 32]), 1) + .unwrap(), + ) + .await + .unwrap(); + } + let scopes = [scope(one.cell_id()), scope(two.cell_id())]; + let grouped = locate_scope_tenants(&layout, &scopes, 2, 2).await.unwrap(); + assert_eq!(grouped.len(), 2); + assert_eq!(grouped[one.tenant().as_bytes()][0], scopes[0]); + assert_eq!(grouped[two.tenant().as_bytes()][0], scopes[1]); + assert!(locate_scope_tenants(&layout, &scopes, 1, 2).await.is_err()); + let missing = CellTarget::new( + TenantId::from_bytes([3; 16]), + APPLICATION, + NAMESPACE, + b"missing", + ) + .unwrap(); + assert!( + locate_scope_tenants(&layout, &[scope(missing.cell_id())], 2, 1) + .await + .is_err() + ); + } + + #[test] + fn catalog_head_parser_rejects_noncanonical_paths() { + let prefix = "root/catalog/tenants/"; + let canonical = format!("{prefix}{}/ab/head.json", "01".repeat(16)); + assert_eq!( + parse_catalog_head(prefix, &canonical).unwrap(), + ([1; 16], 0xab) + ); + assert!(parse_catalog_head(prefix, &canonical.replace("/ab/", "/AB/")).is_err()); + assert!(parse_catalog_head(prefix, &canonical.replace("head.json", "tail.json")).is_err()); + } + + struct EmptySealedFollower; + + impl NodeLogTransport for EmptySealedFollower { + fn append<'a>( + &'a self, + _member: NodeId, + _request: AppendRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async { Err(Error::Node("test follower cannot append")) }) + } + + fn seal<'a>( + &'a self, + _member: NodeId, + _request: SealRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async { + Ok(FollowerReceipt { + base_sequence: 0, + durable_through: 0, + }) + }) + } + + fn retire<'a>( + &'a self, + _member: NodeId, + _request: RetireRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async { Err(Error::Node("test follower cannot retire")) }) + } + + fn tail<'a>( + &'a self, + _member: NodeId, + _request: TailRequest, + ) -> BoxFuture<'a, Result>> { + Box::pin(async { Ok(Vec::new()) }) + } + } + + fn node_advertisement( + node: NodeId, + session: SessionId, + key: u8, + follower_bytes: u64, + now: i64, + lease_ms: i64, + ) -> NodeAdvertisement { + NodeAdvertisement::sign( + node, + session, + format!("https://node-{key}.internal:8081"), + Digest::from_bytes([80; 32]), + Digest::from_bytes([key; 32]), + Digest::from_bytes([81; 32]), + Digest::from_bytes([82; 32]), + &SigningKey::from_bytes(&[key; 32]), + 1, + now, + now + lease_ms, + vec![Digest::from_bytes([86; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 16 << 20, + free_disk_bytes: 1 << 30, + follower_free_bytes: follower_bytes, + job_credits: 8, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + ..NodeCapacity::default() + }, + ) + .unwrap() + } + + #[tokio::test] + async fn empty_follower_tail_seals_claim_before_takeover() { + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("node-log-recovery-empty-tail"), + *APPLICATION_ID.as_bytes(), + ); + let directory = NodeDirectory::new( + layout.clone(), + Digest::from_bytes([80; 32]), + Digest::from_bytes([81; 32]), + Digest::from_bytes([82; 32]), + ); + let follower_node = NodeId::from_bytes([1; 16]); + let follower_session = SessionId::from_bytes([2; 16]); + let leader_node = NodeId::from_bytes([3; 16]); + let leader_session = SessionId::from_bytes([4; 16]); + let claimant_node = NodeId::from_bytes([5; 16]); + let claimant_session = SessionId::from_bytes([6; 16]); + let now = unix_time_ms().unwrap(); + directory + .create( + node_advertisement(follower_node, follower_session, 1, 1 << 30, now, 15_000), + now, + ) + .await + .unwrap(); + let leader = directory + .create( + node_advertisement(leader_node, leader_session, 3, 0, now, 3_000), + now, + ) + .await + .unwrap(); + let enrolled = directory + .recruit_log(&leader, 1, 4_096, 16, now) + .await + .unwrap(); + directory.activate_log(&enrolled, now).await.unwrap(); + directory + .create( + node_advertisement(claimant_node, claimant_session, 5, 0, now, 15_000), + now, + ) + .await + .unwrap(); + tokio::time::sleep(Duration::from_millis(3_100)).await; + let fenced = directory + .claim_expired_for_recovery(leader_session, claimant_session, unix_time_ms().unwrap()) + .await + .unwrap(); + let scratch = tempfile::tempdir().unwrap(); + let result = recover_fenced_node_log( + &directory, + &layout, + Arc::new(EmptySealedFollower), + fenced, + Limits::default(), + scratch.path().to_owned(), + 1, + 1, + ) + .await + .unwrap(); + assert!(result.controls.is_empty()); + assert_eq!(result.takeover.claimant(), claimant_session); + assert!( + directory + .takeover_proof(leader_session, claimant_session, unix_time_ms().unwrap()) + .await + .unwrap() + .is_some() + ); + } + + #[tokio::test] + async fn untiered_account_frame_is_pinned_and_restored_before_takeover() { + let application = Arc::new( + Beyonddb::compile(BuildDescriptor { + source_revision: "node-log-recovery-test".into(), + cargo_lock_digest: Digest::from_bytes([7; 32]), + }) + .unwrap(), + ); + let target = account_target("123456789012").unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("node-log-recovery-account-frame"), + *APPLICATION_ID.as_bytes(), + ); + let files = tempfile::tempdir().unwrap(); + let source = CellNodeBuilder::new(Arc::clone(&application)) + .with_runtime(SqlWorkerPool::new(1, 8).unwrap(), 16 << 20) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(1 << 30))) + .with_session(SessionId::from_bytes([4; 16])) + .build_unleased_for_maintenance() + .unwrap(); + let catalog = CellCatalog::new(layout.clone(), target.tenant()); + let proof = catalog + .provision( + CatalogEntry::new( + &target, + CatalogRole::Sql, + application + .registry() + .module_code("beyonddb-account") + .unwrap(), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let authority = CellAuthority::new(layout.clone()); + let leader_session = SessionId::from_bytes([4; 16]); + let incarnation = IncarnationId::from_bytes([8; 16]); + let initial = authority + .create_initial( + &proof, + incarnation, + Owner { + session: leader_session, + endpoint: "https://leader.internal:8081".into(), + }, + ) + .await + .unwrap(); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + Limits::default(), + ) + .unwrap(); + let handle = source + .runtime() + .bootstrap( + proof.clone(), + replica.clone(), + authority.clone(), + initial, + files.path().join("leader.sqlite"), + initialize_account, + ) + .await + .unwrap(); + drop(handle); + let observed = authority.load(target.cell_id()).await.unwrap().unwrap(); + let predecessor = observed.value().ltx_root().unwrap(); + let tail_path = files.path().join("tail.sqlite"); + let writable = replica + .open_root(&predecessor) + .await + .unwrap() + .paged() + .prepare_writable(&tail_path) + .await + .unwrap(); + let mut writer = writable.open_writable(&tail_path).unwrap(); + writer + .transaction(|transaction| { + transaction.execute( + "UPDATE sys_meta SET commit_sequence = commit_sequence + 1, logical_time_ms = logical_time_ms + 1 WHERE singleton = 1", + [], + )?; + Ok(()) + }) + .unwrap(); + let capture = writer.capture().unwrap(); + let frames = capture + .segments + .iter() + .enumerate() + .map(|(index, segment)| { + encode_node_frame( + NodeFrameScope { + leader_session: *leader_session.as_bytes(), + log_epoch: 1, + node_sequence: u64::try_from(index).unwrap() + 1, + application: *APPLICATION_ID.as_bytes(), + cell: *target.cell_id().as_bytes(), + incarnation: *incarnation.as_bytes(), + cell_epoch: observed.value().epoch, + commit_sequence: predecessor.commit_sequence + 1, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + Limits::default(), + ) + .unwrap() + .encoded() + .clone() + }) + .collect::>(); + assert!(!frames.is_empty()); + writer.close().unwrap(); + + let directory = NodeDirectory::new( + layout.clone(), + Digest::from_bytes([80; 32]), + Digest::from_bytes([81; 32]), + Digest::from_bytes([82; 32]), + ); + let follower_node = NodeId::from_bytes([1; 16]); + let follower_session = SessionId::from_bytes([2; 16]); + let leader_node = NodeId::from_bytes([3; 16]); + let claimant_session = SessionId::from_bytes([6; 16]); + let now = unix_time_ms().unwrap(); + directory + .create( + node_advertisement(follower_node, follower_session, 1, 1 << 30, now, 15_000), + now, + ) + .await + .unwrap(); + let leader = directory + .create( + node_advertisement(leader_node, leader_session, 3, 0, now, 3_000), + now, + ) + .await + .unwrap(); + let enrolled = directory + .recruit_log(&leader, 1, 4_096, 16, now) + .await + .unwrap(); + directory.activate_log(&enrolled, now).await.unwrap(); + directory + .create( + node_advertisement( + NodeId::from_bytes([5; 16]), + claimant_session, + 5, + 0, + now, + 15_000, + ), + now, + ) + .await + .unwrap(); + let follower = cellule_runtime::FollowerStore::open( + files.path().join("follower"), + Limits::default(), + DiskBudget::new(1 << 30), + ) + .unwrap(); + let transport: Arc = + Arc::new(LocalFollowerTransport::new(follower_node, follower)); + transport + .append( + follower_node, + AppendRequest { + leader_session, + log_epoch: 1, + frames, + covered_through: 0, + }, + ) + .await + .unwrap(); + tokio::time::sleep(Duration::from_millis(3_100)).await; + let successor = cellule_runtime::CellRuntime::new( + SqlWorkerPool::new(1, 8).unwrap(), + 16 << 20, + claimant_session, + ) + .unwrap(); + let provisioner = CellInitialPartitionProvisioner::new( + successor.clone(), + application, + layout.clone(), + claimant_session, + "https://successor.internal:8081".into(), + files.path().join("recovery"), + ) + .unwrap() + .with_test_node_log_recovery(transport); + let restored = provisioner + .takeover_expired_account("123456789012", &directory) + .await + .unwrap(); + assert!( + directory + .takeover_proof(leader_session, claimant_session, unix_time_ms().unwrap()) + .await + .unwrap() + .is_some() + ); + let replayed_sequence = restored + .query(64, 64, |connection| { + let sequence = connection.query_row( + "SELECT commit_sequence FROM sys_meta WHERE singleton = 1", + [], + |row| row.get::<_, i64>(0), + )?; + Ok(sequence.to_be_bytes().to_vec()) + }) + .await + .unwrap(); + assert_eq!( + replayed_sequence, + (predecessor.commit_sequence as i64 + 1).to_be_bytes() + ); + assert_eq!( + authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .root + .as_ref() + .unwrap() + .commit_sequence, + predecessor.commit_sequence + 1 + ); + restored.drain().await.unwrap(); + successor.shutdown().await.unwrap(); + } +} diff --git a/src/server/node_log_sender.rs b/src/server/node_log_sender.rs new file mode 100644 index 0000000..5ef76a8 --- /dev/null +++ b/src/server/node_log_sender.rs @@ -0,0 +1,1036 @@ +//! Pinned mTLS transport for Cellule's follower-log operations. + +use std::{ + collections::VecDeque, + sync::{Arc, Mutex}, + time::Duration, +}; + +use axum::body::Bytes; +use cellule_peer_http::PeerTlsClient; +use cellule_runtime::{ + Error, NodeLeaseGuard, Result, + follower::{FollowerReceipt, FollowerStore, FollowerTailPage}, + identity::{Digest, NodeId, SessionId}, + node::{ + NodeAdvertisement, NodeDirectory, + log_transport::{AppendRequest, NodeLogTransport, RetireRequest, SealRequest, TailRequest}, + }, +}; +use futures_util::{StreamExt, future::BoxFuture}; +use reqwest::{StatusCode, Url, header}; + +use super::{ + node_lease::unix_time_ms, + node_log_receiver::{self, Operation, WireRequest}, +}; + +const PEER_CACHE_MS: i64 = 500; +const MAX_CACHED_PEERS: usize = 128; +const RESPONSE_TIMEOUT: Duration = Duration::from_secs(32); +const MAX_TOTAL_TAIL_BYTES: usize = 64 * 1024 * 1024; +const MAX_TAIL_PAGES: usize = 4096; + +/// Outbound follower-log transport bound to one live node session. +/// +/// The receiver rechecks directory authority for every operation. A short +/// sender cache avoids resolving the same enrolled member for each commit. +#[derive(Clone)] +pub struct PeerNodeLogTransport { + directory: NodeDirectory, + tls: PeerTlsClient, + session: SessionId, + node: NodeId, + guard: NodeLeaseGuard, + local_store: Option>, + peers: Arc>>, +} + +#[derive(Clone)] +struct CachedPeer { + member: NodeId, + session: SessionId, + certificate: Digest, + public_key: [u8; 32], + verified_at_ms: i64, + expires_at_ms: i64, + endpoint: Url, + client: reqwest::Client, +} + +impl PeerNodeLogTransport { + /// Return the boot session whose lease and TLS identity bind this transport. + pub const fn session(&self) -> SessionId { + self.session + } + + pub(super) const fn node(&self) -> NodeId { + self.node + } + + /// Creates a transport that authenticates each follower against the fleet directory. + #[must_use] + pub fn new( + directory: NodeDirectory, + tls: PeerTlsClient, + session: SessionId, + node: NodeId, + guard: NodeLeaseGuard, + ) -> Self { + Self { + directory, + tls, + session, + node, + guard, + local_store: None, + peers: Arc::new(Mutex::new(VecDeque::new())), + } + } + + /// Allow a recovery claimant to seal/read its own persistent follower lane. + /// + /// Local recovery still requires the directory's fenced-owner claim. + #[must_use] + pub fn with_local_follower_store(mut self, store: Arc) -> Self { + self.local_store = Some(store); + self + } + + async fn local_recovery_store( + &self, + member: NodeId, + leader: SessionId, + epoch: u64, + ) -> Result<&FollowerStore> { + self.guard.check()?; + if member != self.node { + return Err(Error::PeerAuthorization("local follower member differs")); + } + let store = self + .local_store + .as_deref() + .ok_or(Error::Peer("local follower store is unavailable"))?; + self.directory + .authorize_log_recovery(leader, self.session, member, epoch, unix_time_ms()?) + .await?; + self.guard.check()?; + Ok(store) + } + + async fn peer(&self, member: NodeId) -> Result { + self.guard.check()?; + if member == self.node { + return Err(Error::PeerAuthorization("follower is the local node")); + } + let now_ms = unix_time_ms()?; + if let Some(peer) = self + .peers + .lock() + .map_err(|_| Error::Peer("node-log peer cache is poisoned"))? + .iter() + .find(|peer| { + peer.member == member + && now_ms.saturating_sub(peer.verified_at_ms) < PEER_CACHE_MS + && now_ms < peer.expires_at_ms + }) + .cloned() + { + return Ok(peer); + } + let advertisement = self + .directory + .resolve_node(member, now_ms) + .await? + .ok_or(Error::Peer("follower has no live advertisement"))?; + self.guard.check()?; + self.refresh_peer(member, advertisement, now_ms) + } + + fn refresh_peer( + &self, + member: NodeId, + advertisement: NodeAdvertisement, + now_ms: i64, + ) -> Result { + let certificate = advertisement.certificate(); + let public_key = advertisement.verifying_key()?.to_bytes(); + let endpoint = Url::parse(advertisement.endpoint()).map_err(transport_error)?; + if endpoint.scheme() != "https" || endpoint.host_str().is_none() { + return Err(Error::PeerAuthorization("follower endpoint is not HTTPS")); + } + let mut peers = self + .peers + .lock() + .map_err(|_| Error::Peer("node-log peer cache is poisoned"))?; + let previous = peers.iter().find(|peer| { + peer.member == member + && peer.session == advertisement.session() + && peer.certificate == certificate + && peer.public_key == public_key + && peer.endpoint == endpoint + }); + let client = match previous { + Some(peer) => peer.client.clone(), + None => self + .tls + .client(certificate, public_key) + .map_err(transport_error)?, + }; + let peer = CachedPeer { + member, + session: advertisement.session(), + certificate, + public_key, + verified_at_ms: now_ms, + expires_at_ms: advertisement.expires_at_ms(), + endpoint, + client, + }; + peers.retain(|cached| cached.member != member); + if peers.len() == MAX_CACHED_PEERS { + peers.pop_front(); + } + peers.push_back(peer.clone()); + Ok(peer) + } + + async fn send(&self, member: NodeId, request: WireRequest, tail: bool) -> Result { + let encoded = request + .encode() + .map_err(|()| Error::Peer("invalid node-log request"))?; + let peer = self.peer(member).await?; + let url = peer + .endpoint + .join("internal/node-log/v1") + .map_err(transport_error)?; + self.guard.check()?; + let response = peer + .client + .post(url) + .header(header::CONTENT_TYPE, node_log_receiver::MEDIA_TYPE) + .header(header::CACHE_CONTROL, "no-store") + .header(header::CONTENT_LENGTH, encoded.len()) + .timeout(RESPONSE_TIMEOUT) + .body(encoded) + .send() + .await + .map_err(|source| { + if source.is_connect() { + transport_error(source) + } else { + unknown_error(source) + } + })?; + self.guard.check()?; + match response.status() { + StatusCode::OK => {} + StatusCode::FORBIDDEN | StatusCode::UNAUTHORIZED => { + return Err(Error::PeerAuthorization( + "follower rejected the peer identity", + )); + } + StatusCode::CONFLICT => { + return Err(Error::Node("follower rejected the log authority")); + } + StatusCode::BAD_REQUEST + | StatusCode::UNSUPPORTED_MEDIA_TYPE + | StatusCode::LENGTH_REQUIRED => { + return Err(Error::Peer("follower rejected the node-log request")); + } + status if status.is_server_error() => { + return Err(Error::PeerTransportUnknown { + context: "follower may have accepted the node-log operation", + source: Box::new(Error::Peer("follower returned a server error")), + }); + } + _ => return Err(Error::Peer("follower rejected the node-log request")), + } + if response + .headers() + .get(header::CONTENT_TYPE) + .and_then(|value| value.to_str().ok()) + != Some(node_log_receiver::MEDIA_TYPE) + { + return Err(unknown_error(Error::Peer( + "follower response type is invalid", + ))); + } + let limit = if tail { + node_log_receiver::MAX_RESPONSE_BYTES + } else { + 21 + }; + if response + .content_length() + .is_some_and(|length| length > limit as u64) + { + return Err(unknown_error(Error::Peer( + "follower response exceeds the byte limit", + ))); + } + let mut body = Vec::new(); + let mut stream = response.bytes_stream(); + while let Some(chunk) = stream.next().await { + let chunk = chunk.map_err(unknown_error)?; + if body.len().saturating_add(chunk.len()) > limit { + return Err(unknown_error(Error::Peer( + "follower response exceeds the byte limit", + ))); + } + body.extend_from_slice(&chunk); + } + self.guard.check()?; + Ok(Bytes::from(body)) + } + + async fn receipt(&self, member: NodeId, request: WireRequest) -> Result { + let response = self.send(member, request, false).await?; + node_log_receiver::decode_receipt(response) + .map_err(|()| unknown_error(Error::Peer("follower receipt is invalid"))) + } +} + +impl NodeLogTransport for PeerNodeLogTransport { + fn append<'a>( + &'a self, + member: NodeId, + request: AppendRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + if request.leader_session != self.session { + return Err(Error::PeerAuthorization( + "append leader is not the local session", + )); + } + self.receipt( + member, + WireRequest { + caller: self.session, + leader: request.leader_session, + epoch: request.log_epoch, + argument: request.covered_through, + operation: Operation::Append(request.frames), + }, + ) + .await + }) + } + + fn seal<'a>( + &'a self, + member: NodeId, + request: SealRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + if member == self.node { + let store = self + .local_recovery_store(member, request.leader_session, request.log_epoch) + .await?; + let receipt = store + .seal(request.leader_session, request.log_epoch) + .await?; + self.guard.check()?; + return Ok(receipt); + } + self.receipt( + member, + WireRequest { + caller: self.session, + leader: request.leader_session, + epoch: request.log_epoch, + argument: 0, + operation: Operation::Seal, + }, + ) + .await + }) + } + + fn retire<'a>( + &'a self, + member: NodeId, + request: RetireRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + if request.leader_session != self.session { + return Err(Error::PeerAuthorization( + "retire leader is not the local session", + )); + } + self.receipt( + member, + WireRequest { + caller: self.session, + leader: request.leader_session, + epoch: request.log_epoch, + argument: request.covered_through, + operation: Operation::Retire, + }, + ) + .await + }) + } + + fn tail<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, Result>> { + Box::pin(async move { + let mut first_sequence = request.first_sequence; + let mut frames = Vec::new(); + let mut total = 0_usize; + for _ in 0..MAX_TAIL_PAGES { + let page = self + .tail_page( + member, + TailRequest { + first_sequence, + ..request + }, + ) + .await?; + for frame in page.frames { + total = total + .checked_add(frame.len()) + .ok_or(Error::Capacity("follower tail exceeds byte limit"))?; + if total > MAX_TOTAL_TAIL_BYTES { + return Err(Error::Capacity("follower tail exceeds byte limit")); + } + frames.push(frame); + } + match page.next_sequence { + Some(next) if next > first_sequence => first_sequence = next, + Some(_) => return Err(Error::Peer("follower tail did not advance")), + None => return Ok(frames), + } + } + Err(Error::Capacity("follower tail exceeds page limit")) + }) + } + + fn tail_page<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + if member == self.node { + let store = self + .local_recovery_store(member, request.leader_session, request.log_epoch) + .await?; + let page = store + .read_tail_page( + request.leader_session, + request.log_epoch, + request.first_sequence, + ) + .await?; + self.guard.check()?; + return Ok(page); + } + let response = self + .send( + member, + WireRequest { + caller: self.session, + leader: request.leader_session, + epoch: request.log_epoch, + argument: request.first_sequence, + operation: Operation::Tail, + }, + true, + ) + .await?; + let page = node_log_receiver::decode_page(response) + .map_err(|()| unknown_error(Error::Peer("follower tail page is invalid")))?; + if let Some(next) = page.next_sequence { + let count = u64::try_from(page.frames.len()) + .map_err(|_| Error::Peer("follower tail frame count overflow"))?; + let expected = request + .first_sequence + .checked_add(count) + .ok_or(Error::Peer("follower tail sequence overflow"))?; + if count == 0 || next != expected { + return Err(unknown_error(Error::Peer( + "follower tail continuation is invalid", + ))); + } + } + Ok(page) + }) + } +} + +fn transport_error(source: impl std::error::Error + Send + Sync + 'static) -> Error { + Error::PeerTransport { + context: "follower HTTP transport failed before acceptance", + source: Box::new(source), + } +} + +fn unknown_error(source: impl std::error::Error + Send + Sync + 'static) -> Error { + Error::PeerTransportUnknown { + context: "follower HTTP reply was lost or invalid", + source: Box::new(source), + } +} + +#[cfg(all(test, unix))] +mod tests { + use super::*; + use std::{ + path::{Path, PathBuf}, + process::Command, + sync::Arc, + }; + + use cellule_ltx::{Db, NodeFrameScope, encode_node_frame}; + use cellule_peer_http::{LoadedPeerTls, PeerTlsIdentity}; + use cellule_runtime::{ + SqlWorkerPool, + cell::actor::CellRuntime, + follower::FollowerStore, + ltx::{CellStorageLayout, DiskBudget, Host, Limits}, + node::{ + NODE_LOG_PROTOCOL_VERSION, NodeCapacity, NodeFailureDomain, + log_recovery::NodeLogRecovery, + }, + }; + use cellule_store::Store; + use object_store::{memory::InMemory, path::Path as ObjectPath}; + + fn run(command: &mut Command) { + assert!(command.output().unwrap().status.success()); + } + + fn certificate(root: &Path, name: &str, ca: &Path, ca_key: &Path) -> (PathBuf, PathBuf) { + let key = root.join(format!("{name}.key")); + let csr = root.join(format!("{name}.csr")); + let cert = root.join(format!("{name}.crt")); + let extension = root.join(format!("{name}.ext")); + run(Command::new("openssl") + .args(["genpkey", "-algorithm", "ED25519", "-out"]) + .arg(&key)); + run(Command::new("openssl") + .args(["req", "-new", "-subj", "/CN=localhost", "-key"]) + .arg(&key) + .arg("-out") + .arg(&csr)); + std::fs::write(&extension, "basicConstraints=critical,CA:FALSE\nkeyUsage=critical,digitalSignature\nextendedKeyUsage=serverAuth,clientAuth\nsubjectAltName=DNS:localhost\n").unwrap(); + run(Command::new("openssl") + .args(["x509", "-req", "-days", "1", "-CAcreateserial", "-in"]) + .arg(&csr) + .arg("-CA") + .arg(ca) + .arg("-CAkey") + .arg(ca_key) + .arg("-extfile") + .arg(&extension) + .arg("-out") + .arg(&cert)); + (cert, key) + } + + fn frame(limits: Limits, leader: SessionId) -> Bytes { + let source = tempfile::TempDir::new().unwrap(); + let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); + database.transaction(|transaction| transaction.execute_batch("CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL); INSERT INTO events(body) VALUES ('one')")).unwrap(); + let capture = database.capture().unwrap(); + let segment = capture.segments.first().unwrap(); + encode_node_frame( + NodeFrameScope { + leader_session: *leader.as_bytes(), + log_epoch: 2, + node_sequence: 1, + application: [3; 16], + cell: [4; 32], + incarnation: [5; 16], + cell_epoch: 6, + commit_sequence: 1, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + limits, + ) + .unwrap() + .encoded() + .clone() + } + + fn advertisement( + node: NodeId, + session: SessionId, + endpoint: String, + tls: &LoadedPeerTls, + follower: bool, + log_capable: bool, + now_ms: i64, + lease_ms: i64, + ) -> NodeAdvertisement { + NodeAdvertisement::sign( + node, + session, + endpoint, + tls.fleet(), + tls.certificate(), + Digest::from_bytes([81; 32]), + Digest::from_bytes([82; 32]), + tls.signing_key(), + 1, + now_ms, + now_ms + lease_ms, + vec![Digest::from_bytes([86; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 16 << 20, + free_disk_bytes: 1 << 30, + follower_free_bytes: if follower { 1 << 30 } else { 0 }, + job_credits: 8, + log_protocol: if log_capable { + NODE_LOG_PROTOCOL_VERSION + } else { + 0 + }, + ..NodeCapacity::default() + }, + ) + .unwrap() + } + + #[tokio::test] + async fn pinned_mtls_append_survives_follower_store_reopen() { + let root = tempfile::TempDir::new().unwrap(); + let ca_key = root.path().join("ca.key"); + let ca = root.path().join("ca.crt"); + run(Command::new("openssl") + .args(["genpkey", "-algorithm", "ED25519", "-out"]) + .arg(&ca_key)); + run(Command::new("openssl") + .args([ + "req", + "-x509", + "-new", + "-days", + "1", + "-subj", + "/CN=BeyondDB Test CA", + "-addext", + "basicConstraints=critical,CA:TRUE", + "-key", + ]) + .arg(&ca_key) + .arg("-out") + .arg(&ca)); + let (leader_cert, leader_key) = certificate(root.path(), "leader", &ca, &ca_key); + let (follower_cert, follower_key) = certificate(root.path(), "follower", &ca, &ca_key); + let leader_tls = LoadedPeerTls::load(&leader_cert, &leader_key, &ca, "localhost").unwrap(); + let follower_tls = + LoadedPeerTls::load(&follower_cert, &follower_key, &ca, "localhost").unwrap(); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let follower_endpoint = format!("https://{}", listener.local_addr().unwrap()); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("follower-transport-test"), + [42; 16], + ); + let directory = NodeDirectory::new( + layout, + leader_tls.fleet(), + Digest::from_bytes([81; 32]), + Digest::from_bytes([82; 32]), + ); + let leader_session = SessionId::from_bytes([1; 16]); + let follower_session = SessionId::from_bytes([2; 16]); + let leader_node = NodeId::from_bytes([3; 16]); + let follower_node = NodeId::from_bytes([4; 16]); + let now_ms = unix_time_ms().unwrap(); + directory + .create( + advertisement( + follower_node, + follower_session, + follower_endpoint, + &follower_tls, + true, + true, + now_ms, + 15_000, + ), + now_ms, + ) + .await + .unwrap(); + let enrolled = directory + .create( + advertisement( + leader_node, + leader_session, + "https://leader.internal:8081".into(), + &leader_tls, + false, + false, + now_ms, + 3_000, + ), + now_ms, + ) + .await + .unwrap(); + let enrolled = directory + .try_recruit_log(&enrolled, 2, 4096, 16, now_ms) + .await + .unwrap() + .unwrap(); + assert_eq!( + enrolled.advertisement().log().unwrap().members(), + &[follower_node] + ); + + let limits = Limits::default(); + let store_root = root.path().join("follower-store"); + let store = Arc::new( + FollowerStore::open(store_root.clone(), limits, DiskBudget::new(1 << 30)).unwrap(), + ); + let runtime = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(1, 8).unwrap(), + 64 << 20, + follower_session, + Host::default(), + ) + .unwrap(); + let guard = NodeLeaseGuard::new(now_ms, now_ms + 15_000).unwrap(); + let receiver = node_log_receiver::FollowerEndpoint::new( + directory.clone(), + runtime, + follower_node, + store.clone(), + guard.clone(), + ); + let follower_client = follower_tls.client_identity(); + let server = tokio::spawn(async move { + axum::serve( + follower_tls.listener(listener), + node_log_receiver::router(receiver) + .into_make_service_with_connect_info::(), + ) + .await + }); + let transport = PeerNodeLogTransport::new( + directory.clone(), + leader_tls.client_identity(), + leader_session, + leader_node, + guard.clone(), + ); + let saved = frame(limits, leader_session); + for _ in 0..2 { + let receipt = transport + .append( + follower_node, + AppendRequest { + leader_session, + log_epoch: 2, + frames: vec![saved.clone()], + covered_through: 0, + }, + ) + .await + .unwrap(); + assert_eq!(receipt.durable_through, 1); + } + let local = PeerNodeLogTransport::new( + directory.clone(), + follower_client, + follower_session, + follower_node, + guard, + ) + .with_local_follower_store(store.clone()); + assert!( + local + .seal( + follower_node, + SealRequest { + leader_session, + log_epoch: 2, + }, + ) + .await + .is_err() + ); + tokio::time::sleep(Duration::from_millis(3_100)).await; + let fenced = directory + .claim_expired_for_recovery(leader_session, follower_session, unix_time_ms().unwrap()) + .await + .unwrap(); + let recovery_transport: Arc = Arc::new(local.clone()); + let recovery = NodeLogRecovery::from_fenced(recovery_transport, &fenced, limits) + .unwrap() + .with_recovery_scratch(root.path().to_owned()); + let sealed = recovery.ensure_sealed_bounded().await.unwrap(); + assert_eq!(sealed.frame_count(), 1); + assert_eq!(sealed.scopes(limits).unwrap()[0].cell, [4; 32]); + assert_eq!( + local + .seal( + follower_node, + SealRequest { + leader_session, + log_epoch: 2, + }, + ) + .await + .unwrap() + .durable_through, + 1 + ); + let page = local + .tail_page( + follower_node, + TailRequest { + leader_session, + log_epoch: 2, + first_sequence: 1, + }, + ) + .await + .unwrap(); + assert_eq!(page.frames, vec![saved.clone()]); + assert_eq!(page.next_sequence, None); + server.abort(); + let _ = server.await; + drop(transport); + drop(local); + drop(store); + let reopened = FollowerStore::open(store_root, limits, DiskBudget::new(1 << 30)).unwrap(); + assert_eq!( + reopened + .seal(leader_session, 2) + .await + .unwrap() + .durable_through, + 1 + ); + assert_eq!( + reopened.read_tail(leader_session, 2, 1).await.unwrap(), + vec![saved] + ); + } + + #[tokio::test] + async fn remote_claimant_recovers_persisted_follower_tail() { + let root = tempfile::TempDir::new().unwrap(); + let ca_key = root.path().join("ca.key"); + let ca = root.path().join("ca.crt"); + run(Command::new("openssl") + .args(["genpkey", "-algorithm", "ED25519", "-out"]) + .arg(&ca_key)); + run(Command::new("openssl") + .args([ + "req", + "-x509", + "-new", + "-days", + "1", + "-subj", + "/CN=BeyondDB Test CA", + "-addext", + "basicConstraints=critical,CA:TRUE", + "-key", + ]) + .arg(&ca_key) + .arg("-out") + .arg(&ca)); + let (leader_cert, leader_key) = certificate(root.path(), "leader", &ca, &ca_key); + let (follower_cert, follower_key) = certificate(root.path(), "follower", &ca, &ca_key); + let (claimant_cert, claimant_key) = certificate(root.path(), "claimant", &ca, &ca_key); + let leader_tls = LoadedPeerTls::load(&leader_cert, &leader_key, &ca, "localhost").unwrap(); + let follower_tls = + LoadedPeerTls::load(&follower_cert, &follower_key, &ca, "localhost").unwrap(); + let claimant_tls = + LoadedPeerTls::load(&claimant_cert, &claimant_key, &ca, "localhost").unwrap(); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let follower_endpoint = format!("https://{}", listener.local_addr().unwrap()); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("remote-follower-recovery-test"), + [42; 16], + ); + let directory = NodeDirectory::new( + layout, + leader_tls.fleet(), + Digest::from_bytes([81; 32]), + Digest::from_bytes([82; 32]), + ); + let leader_session = SessionId::from_bytes([1; 16]); + let follower_session = SessionId::from_bytes([2; 16]); + let claimant_session = SessionId::from_bytes([3; 16]); + let leader_node = NodeId::from_bytes([4; 16]); + let follower_node = NodeId::from_bytes([5; 16]); + let claimant_node = NodeId::from_bytes([6; 16]); + let now_ms = unix_time_ms().unwrap(); + directory + .create( + advertisement( + follower_node, + follower_session, + follower_endpoint, + &follower_tls, + true, + true, + now_ms, + 15_000, + ), + now_ms, + ) + .await + .unwrap(); + let leader = directory + .create( + advertisement( + leader_node, + leader_session, + "https://leader.internal:8081".into(), + &leader_tls, + false, + false, + now_ms, + 3_000, + ), + now_ms, + ) + .await + .unwrap(); + let enrolled = directory + .try_recruit_log(&leader, 2, 4096, 16, now_ms) + .await + .unwrap() + .unwrap(); + assert_eq!( + enrolled.advertisement().log().unwrap().members(), + &[follower_node] + ); + directory + .create( + advertisement( + claimant_node, + claimant_session, + "https://claimant.internal:8081".into(), + &claimant_tls, + false, + true, + now_ms, + 15_000, + ), + now_ms, + ) + .await + .unwrap(); + + let limits = Limits::default(); + let store = Arc::new( + FollowerStore::open( + root.path().join("follower-store"), + limits, + DiskBudget::new(1 << 30), + ) + .unwrap(), + ); + let runtime = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(1, 8).unwrap(), + 64 << 20, + follower_session, + Host::default(), + ) + .unwrap(); + let follower_guard = NodeLeaseGuard::new(now_ms, now_ms + 15_000).unwrap(); + let receiver = node_log_receiver::FollowerEndpoint::new( + directory.clone(), + runtime, + follower_node, + store, + follower_guard, + ); + let server = tokio::spawn(async move { + axum::serve( + follower_tls.listener(listener), + node_log_receiver::router(receiver) + .into_make_service_with_connect_info::(), + ) + .await + }); + let leader_transport = PeerNodeLogTransport::new( + directory.clone(), + leader_tls.client_identity(), + leader_session, + leader_node, + NodeLeaseGuard::new(now_ms, now_ms + 15_000).unwrap(), + ); + let saved = frame(limits, leader_session); + assert_eq!( + leader_transport + .append( + follower_node, + AppendRequest { + leader_session, + log_epoch: 2, + frames: vec![saved.clone()], + covered_through: 0, + }, + ) + .await + .unwrap() + .durable_through, + 1 + ); + let claimant_transport = PeerNodeLogTransport::new( + directory.clone(), + claimant_tls.client_identity(), + claimant_session, + claimant_node, + NodeLeaseGuard::new(now_ms, now_ms + 15_000).unwrap(), + ); + assert!( + claimant_transport + .seal( + follower_node, + SealRequest { + leader_session, + log_epoch: 2, + }, + ) + .await + .is_err() + ); + tokio::time::sleep(Duration::from_millis(3_100)).await; + let fenced = directory + .claim_expired_for_recovery(leader_session, claimant_session, unix_time_ms().unwrap()) + .await + .unwrap(); + let recovery_transport: Arc = Arc::new(claimant_transport.clone()); + let recovery = NodeLogRecovery::from_fenced(recovery_transport, &fenced, limits) + .unwrap() + .with_recovery_scratch(root.path().to_owned()); + let sealed = recovery.ensure_sealed_bounded().await.unwrap(); + assert_eq!(sealed.frame_count(), 1); + assert_eq!(sealed.scopes(limits).unwrap()[0].cell, [4; 32]); + let page = claimant_transport + .tail_page( + follower_node, + TailRequest { + leader_session, + log_epoch: 2, + first_sequence: 1, + }, + ) + .await + .unwrap(); + assert_eq!(page.frames, vec![saved]); + assert_eq!(page.next_sequence, None); + server.abort(); + let _ = server.await; + } +} diff --git a/src/server/peer_receiver.rs b/src/server/peer_receiver.rs index 957176c..34f51de 100644 --- a/src/server/peer_receiver.rs +++ b/src/server/peer_receiver.rs @@ -1,4 +1,5 @@ use std::{ + collections::HashMap, future::Future, pin::Pin, sync::Arc, @@ -20,7 +21,7 @@ use cellule_runtime::cell::{ }; use cellule_runtime::client::LocalCellResolver; use cellule_runtime::control::{ControlState, authority::CellAuthority}; -use cellule_runtime::identity::{CellTarget, Digest, SessionId}; +use cellule_runtime::identity::{CellId, CellTarget, Digest, SessionId}; use cellule_runtime::ltx::CellStorageLayout; use cellule_runtime::node::NodeDirectory; use cellule_runtime::peer::{ @@ -28,6 +29,7 @@ use cellule_runtime::peer::{ }; use cellule_runtime::registry::Registry; use cellule_runtime::{Error, Result}; +use tokio::sync::RwLock; use super::{BeyonddbPeerScope, node_lease::unix_time_ms}; use crate::{DATA_MODULE, DATA_NAMESPACE, MODULE, NAMESPACE, credentials, transaction_coordinator}; @@ -43,20 +45,40 @@ pub(super) struct LocalResolver { runtime: CellRuntime, layout: CellStorageLayout, registry: Arc, + catalog_cache: Arc>>, + handle_cache: Option>>>, provisioner: Option>, placement: Option>, bootstrap: Option, } +const LOCAL_HANDLE_CACHE_TTL: Duration = Duration::from_millis(500); + +#[derive(Clone)] +struct CachedHandle { + handle: CellHandle, + expires_at: Instant, +} + impl LocalResolver { pub(super) fn serving( peers: &super::BeyonddbPeers, provisioner: Arc, + ) -> Self { + Self::serving_with_cache(peers, provisioner, false) + } + + pub(super) fn serving_with_cache( + peers: &super::BeyonddbPeers, + provisioner: Arc, + handle_cache_enabled: bool, ) -> Self { Self { runtime: peers.runtime.clone(), layout: peers.layout.clone(), registry: peers.registry.clone(), + catalog_cache: Arc::new(RwLock::new(HashMap::new())), + handle_cache: handle_cache_enabled.then(|| Arc::new(RwLock::new(HashMap::new()))), provisioner: Some(provisioner), placement: None, bootstrap: None, @@ -80,10 +102,36 @@ impl LocalCellResolver for LocalResolver { let resolver = self.clone(); Box::pin(async move { BeyonddbPeerScope.check_target(&target)?; - let proof = CellCatalog::new(resolver.layout.clone(), target.tenant()) - .lookup(target.cell_id()) - .await? - .ok_or(Error::CellNotActive)?; + let cell = target.cell_id(); + if let Some(cache) = resolver.handle_cache.as_ref() { + let cached = cache.read().await.get(&cell).cloned(); + if let Some(cached) = cached { + if cached.expires_at > Instant::now() { + return Ok(Some(cached.handle)); + } + cache.write().await.remove(&cell); + } + } + let proof = if let Some(proof) = resolver + .catalog_cache + .read() + .await + .get(&target.cell_id()) + .cloned() + { + proof + } else { + let proof = CellCatalog::new(resolver.layout.clone(), target.tenant()) + .lookup(target.cell_id()) + .await? + .ok_or(Error::CellNotActive)?; + resolver + .catalog_cache + .write() + .await + .insert(target.cell_id(), proof.clone()); + proof + }; let module = match target.namespace() { NAMESPACE => MODULE, DATA_NAMESPACE => DATA_MODULE, @@ -118,6 +166,15 @@ impl LocalCellResolver for LocalResolver { .local_handle(proof.clone(), control) .await? { + if let Some(cache) = resolver.handle_cache.as_ref() { + cache.write().await.insert( + cell, + CachedHandle { + handle: local.clone(), + expires_at: Instant::now() + LOCAL_HANDLE_CACHE_TTL, + }, + ); + } return Ok(Some(local)); } let Some(provisioner) = &resolver.provisioner else { @@ -294,6 +351,8 @@ pub(super) fn peer_router( runtime: runtime.clone(), layout: peers.layout.clone(), registry: peers.registry.clone(), + catalog_cache: Arc::new(RwLock::new(HashMap::new())), + handle_cache: None, // The sender selects ownership before forwarding. A receiver may // only dispatch to that active owner; a raced release must reject. provisioner: None, diff --git a/src/transaction_coordinator.rs b/src/transaction_coordinator.rs index b6f07e0..d80552d 100644 --- a/src/transaction_coordinator.rs +++ b/src/transaction_coordinator.rs @@ -37,7 +37,7 @@ static NAMESPACES: [NamespaceDescriptor; 1] = [NamespaceDescriptor { effect_targets: &[], dead_letter: None, }]; -static COMMANDS: [OperationDescriptor; 7] = [ +static COMMANDS: [OperationDescriptor; 10] = [ operation(1), crate::participant::phase_operation(2), operation(3), @@ -45,6 +45,21 @@ static COMMANDS: [OperationDescriptor; 7] = [ crate::transaction_transport::upload_operation(5), crate::participant::phase_operation(6), crate::participant::phase_operation(7), + OperationDescriptor { + input_limit: 64 * 1024, + output_limit: 64 * 1024, + ..operation(8) + }, + OperationDescriptor { + input_limit: 64 * 1024, + output_limit: 64 * 1024, + ..operation(9) + }, + OperationDescriptor { + input_limit: 64 * 1024, + output_limit: 64 * 1024, + ..operation(10) + }, ]; static QUERIES: [OperationDescriptor; 6] = [ OperationDescriptor { @@ -120,10 +135,13 @@ impl cellule_runtime::registry::CellModule for CoordinatorModule { registry.bind_command::>()?; registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_query::()?; registry.bind_query::()?; registry.bind_query::()?; @@ -235,7 +253,7 @@ impl Command for BeginCrossCellTransaction { const MODULE: &'static str = MODULE; const ID: u32 = 1; const CODEC_VERSION: u32 = 1; - type Input = Json; + type Input = Json>; type Output = Json; fn execute( diff --git a/src/transaction_coordinator/phase.rs b/src/transaction_coordinator/phase.rs index f63f55d..8e4d55c 100644 --- a/src/transaction_coordinator/phase.rs +++ b/src/transaction_coordinator/phase.rs @@ -111,48 +111,88 @@ impl Command for RecordParticipantPrepare { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - phase_identity(context, &input.account_id, &input.routing_key)?; - let (state, prepared) = match phase_row(context, &input)? { - PhaseRow::Missing => { - return Ok(CommandResult::Rejected(Json( - CoordinatorPhaseOutcome::Missing, - ))); - } - PhaseRow::WrongParticipant => { - return Ok(CommandResult::Rejected(Json( - CoordinatorPhaseOutcome::WrongParticipant, - ))); - } - PhaseRow::Found { - state, prepared, .. - } => (state, prepared), - }; - if state != 0 { + record_participant_prepare(context, input) + } +} + +fn record_participant_prepare( + context: &mut CommandContext<'_, '_>, + input: CoordinatorPhaseInput, +) -> Result>> { + phase_identity(context, &input.account_id, &input.routing_key)?; + let (state, prepared) = match phase_row(context, &input)? { + PhaseRow::Missing => { return Ok(CommandResult::Rejected(Json( - CoordinatorPhaseOutcome::WrongDecision, + CoordinatorPhaseOutcome::Missing, ))); } - if prepared.is_some() { - return Ok(CommandResult::Success(Json( - CoordinatorPhaseOutcome::Replay, + PhaseRow::WrongParticipant => { + return Ok(CommandResult::Rejected(Json( + CoordinatorPhaseOutcome::WrongParticipant, ))); } - let sequence = i64::try_from(input.sequence) - .ok() - .filter(|value| *value > 0) - .ok_or(Error::Command("invalid prepare sequence"))?; - context.sql(&statement( - "UPDATE ddb_coordinator_participants SET prepared_sequence = ?1 \ + PhaseRow::Found { + state, prepared, .. + } => (state, prepared), + }; + if state != 0 { + return Ok(CommandResult::Rejected(Json( + CoordinatorPhaseOutcome::WrongDecision, + ))); + } + if prepared.is_some() { + return Ok(CommandResult::Success(Json( + CoordinatorPhaseOutcome::Replay, + ))); + } + let sequence = i64::try_from(input.sequence) + .ok() + .filter(|value| *value > 0) + .ok_or(Error::Command("invalid prepare sequence"))?; + context.sql(&statement( + "UPDATE ddb_coordinator_participants SET prepared_sequence = ?1 \ WHERE transaction_id = ?2 AND position = ?3 AND prepared_sequence IS NULL", - vec![ - SqlValue::Integer(sequence), - SqlValue::Blob(input.transaction_id.to_vec()), - SqlValue::Integer(i64::from(input.position)), - ], - ))?; - Ok(CommandResult::Success(Json( - CoordinatorPhaseOutcome::Recorded, - ))) + vec![ + SqlValue::Integer(sequence), + SqlValue::Blob(input.transaction_id.to_vec()), + SqlValue::Integer(i64::from(input.position)), + ], + ))?; + Ok(CommandResult::Success(Json( + CoordinatorPhaseOutcome::Recorded, + ))) +} + +/// Batch durable evidence for independent participant prepares. +pub struct RecordParticipantPrepares; + +impl Command for RecordParticipantPrepares { + const MODULE: &'static str = MODULE; + const ID: u32 = 9; + const CODEC_VERSION: u32 = 1; + type Input = Json>; + type Output = Json>; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + if input.is_empty() || input.len() > 100 { + return Err(Error::Command("prepare evidence batch is outside 1..=100")); + } + let results = input + .into_iter() + .map(|input| record_participant_prepare(context, input)) + .collect::>>()?; + let outcomes = results + .into_iter() + .map(|result| match result { + CommandResult::Success(Json(outcome)) | CommandResult::Rejected(Json(outcome)) => { + outcome + } + }) + .collect(); + Ok(CommandResult::Success(Json(outcomes))) } } @@ -278,82 +318,124 @@ impl Command for RecordParticipantResolution { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - phase_identity(context, &input.account_id, &input.routing_key)?; - let (state, resolved) = match phase_row(context, &input)? { - PhaseRow::Missing => { - return Ok(CommandResult::Rejected(Json( - CoordinatorPhaseOutcome::Missing, - ))); - } - PhaseRow::WrongParticipant => { - return Ok(CommandResult::Rejected(Json( - CoordinatorPhaseOutcome::WrongParticipant, - ))); - } - PhaseRow::Found { - state, resolved, .. - } => (state, resolved), - }; - if state == 0 { + record_participant_resolution(context, input) + } +} + +fn record_participant_resolution( + context: &mut CommandContext<'_, '_>, + input: CoordinatorPhaseInput, +) -> Result>> { + phase_identity(context, &input.account_id, &input.routing_key)?; + let (state, resolved) = match phase_row(context, &input)? { + PhaseRow::Missing => { return Ok(CommandResult::Rejected(Json( - CoordinatorPhaseOutcome::WrongDecision, + CoordinatorPhaseOutcome::Missing, ))); } - if resolved.is_some() { - return Ok(CommandResult::Success(Json( - CoordinatorPhaseOutcome::Replay, + PhaseRow::WrongParticipant => { + return Ok(CommandResult::Rejected(Json( + CoordinatorPhaseOutcome::WrongParticipant, ))); } - let sequence = i64::try_from(input.sequence) - .ok() - .filter(|value| *value > 0) - .ok_or(Error::Command("invalid resolution sequence"))?; - context.sql(&statement( - "UPDATE ddb_coordinator_participants SET resolved_sequence = ?1 \ + PhaseRow::Found { + state, resolved, .. + } => (state, resolved), + }; + if state == 0 { + return Ok(CommandResult::Rejected(Json( + CoordinatorPhaseOutcome::WrongDecision, + ))); + } + if resolved.is_some() { + return Ok(CommandResult::Success(Json( + CoordinatorPhaseOutcome::Replay, + ))); + } + let sequence = i64::try_from(input.sequence) + .ok() + .filter(|value| *value > 0) + .ok_or(Error::Command("invalid resolution sequence"))?; + context.sql(&statement( + "UPDATE ddb_coordinator_participants SET resolved_sequence = ?1 \ WHERE transaction_id = ?2 AND position = ?3 AND resolved_sequence IS NULL", - vec![ - SqlValue::Integer(sequence), - SqlValue::Blob(input.transaction_id.to_vec()), - SqlValue::Integer(i64::from(input.position)), - ], - ))?; - // Terminal write replay needs the digest, target and receipts, but no - // operation images. Committed reads still need their position mapping; - // aborted reads never return images. Keep the decision payload at -1. - context.sql(&statement( - "DELETE FROM ddb_transaction_payloads WHERE transaction_id = ?1 AND position = ?2 \ + vec![ + SqlValue::Integer(sequence), + SqlValue::Blob(input.transaction_id.to_vec()), + SqlValue::Integer(i64::from(input.position)), + ], + ))?; + // Terminal write replay needs the digest, target and receipts, but no + // operation images. Committed reads still need their position mapping; + // aborted reads never return images. Keep the decision payload at -1. + context.sql(&statement( + "DELETE FROM ddb_transaction_payloads WHERE transaction_id = ?1 AND position = ?2 \ AND EXISTS (SELECT 1 FROM ddb_coordinator_participants \ WHERE transaction_id = ?1 AND position = ?2 AND (retain_operations = 0 OR ?3 = 2))", - vec![ - SqlValue::Blob(input.transaction_id.to_vec()), - SqlValue::Integer(i64::from(input.position)), - SqlValue::Integer(state), - ], - ))?; - context.sql(&statement( - "UPDATE ddb_coordinator_participants SET operation_chunks = NULL \ + vec![ + SqlValue::Blob(input.transaction_id.to_vec()), + SqlValue::Integer(i64::from(input.position)), + SqlValue::Integer(state), + ], + ))?; + context.sql(&statement( + "UPDATE ddb_coordinator_participants SET operation_chunks = NULL \ WHERE transaction_id = ?1 AND position = ?2 AND (retain_operations = 0 OR ?3 = 2)", - vec![ - SqlValue::Blob(input.transaction_id.to_vec()), - SqlValue::Integer(i64::from(input.position)), - SqlValue::Integer(state), - ], - ))?; - // ExtendDB rolls back token claims on canceled writes. Release the slot - // only after every abort resolution, so retries cannot race old intents. - context.sql(&statement( - "UPDATE ddb_coordinator_transactions SET unresolved_count = unresolved_count - 1, \ + vec![ + SqlValue::Blob(input.transaction_id.to_vec()), + SqlValue::Integer(i64::from(input.position)), + SqlValue::Integer(state), + ], + ))?; + // ExtendDB rolls back token claims on canceled writes. Release the slot + // only after every abort resolution, so retries cannot race old intents. + context.sql(&statement( + "UPDATE ddb_coordinator_transactions SET unresolved_count = unresolved_count - 1, \ token = CASE WHEN unresolved_count = 1 AND state = 2 THEN NULL ELSE token END, \ completed_at_ms = CASE WHEN unresolved_count = 1 THEN ?2 ELSE completed_at_ms END \ WHERE transaction_id = ?1 AND unresolved_count > 0", - vec![ - SqlValue::Blob(input.transaction_id.to_vec()), - SqlValue::Integer(context.now_ms()), - ], - ))?; - Ok(CommandResult::Success(Json( - CoordinatorPhaseOutcome::Recorded, - ))) + vec![ + SqlValue::Blob(input.transaction_id.to_vec()), + SqlValue::Integer(context.now_ms()), + ], + ))?; + Ok(CommandResult::Success(Json( + CoordinatorPhaseOutcome::Recorded, + ))) +} + +/// Batch durable evidence for independent participant resolutions. +pub struct RecordParticipantResolutions; + +impl Command for RecordParticipantResolutions { + const MODULE: &'static str = MODULE; + const ID: u32 = 10; + const CODEC_VERSION: u32 = 1; + type Input = Json>; + type Output = Json>; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + if input.is_empty() || input.len() > 100 { + return Err(Error::Command( + "resolution evidence batch is outside 1..=100", + )); + } + let results = input + .into_iter() + .map(|input| record_participant_resolution(context, input)) + .collect::>>()?; + let outcomes = results + .into_iter() + .map(|result| match result { + CommandResult::Success(Json(outcome)) | CommandResult::Rejected(Json(outcome)) => { + outcome + } + }) + .collect(); + Ok(CommandResult::Success(Json(outcomes))) } } diff --git a/src/transaction_coordinator/read_release.rs b/src/transaction_coordinator/read_release.rs index 1b1727a..023cb88 100644 --- a/src/transaction_coordinator/read_release.rs +++ b/src/transaction_coordinator/read_release.rs @@ -1,6 +1,7 @@ //! Durable acknowledgement and recovery of consumed transactional read images. use cellule_runtime::registry::{Command, CommandContext, CommandResult}; +use serde::{Deserialize, Serialize}; use super::phase::phase_identity; use super::{ @@ -71,75 +72,145 @@ impl Command for RecordReadResultRelease { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - phase_identity(context, &input.account_id, &input.routing_key)?; - if input.sequence == 0 || input.sequence > i64::MAX as u64 { - return Err(Error::Command("invalid read release sequence")); - } - let rows = context.sql(&statement( - "SELECT p.cell_id, p.operation_chunks, p.retain_operations, p.resolved_sequence, \ + record_read_result_release(context, input) + } +} + +fn record_read_result_release( + context: &mut CommandContext<'_, '_>, + input: CoordinatorPhaseInput, +) -> Result>> { + phase_identity(context, &input.account_id, &input.routing_key)?; + if input.sequence == 0 || input.sequence > i64::MAX as u64 { + return Err(Error::Command("invalid read release sequence")); + } + let rows = context.sql(&statement( + "SELECT p.cell_id, p.operation_chunks, p.retain_operations, p.resolved_sequence, \ t.state, t.unresolved_count, t.read_release_count FROM ddb_coordinator_participants p \ JOIN ddb_coordinator_transactions t ON t.transaction_id = p.transaction_id \ WHERE p.transaction_id = ?1 AND p.position = ?2 AND t.account_id = ?3", - vec![ - SqlValue::Blob(input.transaction_id.to_vec()), - SqlValue::Integer(i64::from(input.position)), - SqlValue::Text(input.account_id), - ], - ))?; - let Some(row) = rows[0].rows.first() else { - return Ok(CommandResult::Rejected(Json( - CoordinatorPhaseOutcome::Missing, - ))); - }; - let [ - SqlValue::Blob(cell), - chunks, - SqlValue::Integer(retain), - resolved, - SqlValue::Integer(state), - SqlValue::Integer(unresolved), - SqlValue::Integer(releases), - ] = row.as_slice() - else { - return Err(Error::Command("invalid read release record")); - }; - if cell.as_slice() != input.participant_cell { - return Ok(CommandResult::Rejected(Json( - CoordinatorPhaseOutcome::WrongParticipant, - ))); - } - if *state != 1 || *unresolved != 0 || *retain != 1 || *resolved == SqlValue::Null { - return Ok(CommandResult::Rejected(Json( - CoordinatorPhaseOutcome::WrongDecision, - ))); - } - if *chunks == SqlValue::Null { - return Ok(CommandResult::Success(Json( - CoordinatorPhaseOutcome::Replay, - ))); - } - if *releases == 0 { - return Ok(CommandResult::Rejected(Json( - CoordinatorPhaseOutcome::WrongDecision, - ))); - } - context.sql(&statement( - "DELETE FROM ddb_transaction_payloads WHERE transaction_id = ?1 AND position = ?2", - vec![ - SqlValue::Blob(input.transaction_id.to_vec()), - SqlValue::Integer(i64::from(input.position)), - ], - ))?; - context.sql(&statement( + vec![ + SqlValue::Blob(input.transaction_id.to_vec()), + SqlValue::Integer(i64::from(input.position)), + SqlValue::Text(input.account_id), + ], + ))?; + let Some(row) = rows[0].rows.first() else { + return Ok(CommandResult::Rejected(Json( + CoordinatorPhaseOutcome::Missing, + ))); + }; + let [ + SqlValue::Blob(cell), + chunks, + SqlValue::Integer(retain), + resolved, + SqlValue::Integer(state), + SqlValue::Integer(unresolved), + SqlValue::Integer(releases), + ] = row.as_slice() + else { + return Err(Error::Command("invalid read release record")); + }; + if cell.as_slice() != input.participant_cell { + return Ok(CommandResult::Rejected(Json( + CoordinatorPhaseOutcome::WrongParticipant, + ))); + } + if *state != 1 || *unresolved != 0 || *retain != 1 || *resolved == SqlValue::Null { + return Ok(CommandResult::Rejected(Json( + CoordinatorPhaseOutcome::WrongDecision, + ))); + } + if *chunks == SqlValue::Null { + return Ok(CommandResult::Success(Json( + CoordinatorPhaseOutcome::Replay, + ))); + } + if *releases == 0 { + return Ok(CommandResult::Rejected(Json( + CoordinatorPhaseOutcome::WrongDecision, + ))); + } + context.sql(&statement( + "DELETE FROM ddb_transaction_payloads WHERE transaction_id = ?1 AND position = ?2", + vec![ + SqlValue::Blob(input.transaction_id.to_vec()), + SqlValue::Integer(i64::from(input.position)), + ], + ))?; + context.sql(&statement( "UPDATE ddb_coordinator_participants SET operation_chunks = NULL WHERE transaction_id = ?1 AND position = ?2", vec![SqlValue::Blob(input.transaction_id.to_vec()), SqlValue::Integer(i64::from(input.position))], ))?; - context.sql(&statement( + context.sql(&statement( "UPDATE ddb_coordinator_transactions SET read_release_count = read_release_count - 1 WHERE transaction_id = ?1", vec![SqlValue::Blob(input.transaction_id.to_vec())], ))?; - Ok(CommandResult::Success(Json( - CoordinatorPhaseOutcome::Recorded, - ))) + Ok(CommandResult::Success(Json( + CoordinatorPhaseOutcome::Recorded, + ))) +} + +/// One participant release receipt recorded after its immutable read images are removed. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct ReadResultRelease { + pub position: u8, + pub participant_cell: [u8; 32], + pub sequence: u64, +} + +/// Batch coordinator evidence for a read result release. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct RecordReadResultReleasesInput { + pub account_id: String, + pub transaction_id: [u8; 16], + pub routing_key: Vec, + pub releases: Vec, +} + +/// Record several participant cleanup receipts in one coordinator Cell command. +pub struct RecordReadResultReleases; + +impl Command for RecordReadResultReleases { + const MODULE: &'static str = MODULE; + const ID: u32 = 8; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json>; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + if input.releases.is_empty() || input.releases.len() > 100 { + return Err(Error::Command("read release batch is outside 1..=100")); + } + let results = input + .releases + .into_iter() + .map(|release| { + record_read_result_release( + context, + CoordinatorPhaseInput { + account_id: input.account_id.clone(), + transaction_id: input.transaction_id, + routing_key: input.routing_key.clone(), + position: release.position, + participant_cell: release.participant_cell, + sequence: release.sequence, + }, + ) + }) + .collect::>>()?; + let outcomes = results + .into_iter() + .map(|result| match result { + CommandResult::Success(Json(outcome)) | CommandResult::Rejected(Json(outcome)) => { + outcome + } + }) + .collect(); + Ok(CommandResult::Success(Json(outcomes))) } } diff --git a/src/transaction_transport.rs b/src/transaction_transport.rs index f40d3ae..fbe192d 100644 --- a/src/transaction_transport.rs +++ b/src/transaction_transport.rs @@ -14,6 +14,7 @@ use crate::{Error, Json, Result, SqlValue}; // spend the fixed transfer lifetime on additional durable round trips. pub(crate) const CHUNK_BYTES: usize = 768 * 1024; pub(crate) const MAX_BYTES: usize = 32 * 1024 * 1024; +pub(crate) const INLINE_BYTES: usize = CHUNK_BYTES - 4096; // Runtime mutation and peer authorization permit five minutes of sender skew. // Add that tolerance to the adapter's one-minute absolute upload deadline. const MAX_FUTURE_EXPIRY_MS: i64 = 6 * 60_000; @@ -101,8 +102,19 @@ impl WireValue for TransactionPayloadChunk { } } +/// Transaction phase input carried inline when it fits in one bounded command. +/// Larger inputs continue to use the durable multipart upload path. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +#[serde(untagged)] +pub enum TransactionCommandInput { + Inline(T), + Reference(TransactionPayloadRef), +} + /// Transaction phase whose input is assembled atomically from uploaded pieces. -pub trait MultipartTransactionCommand: Command> { +pub trait MultipartTransactionCommand: + Command>> +{ type Payload: Serialize + DeserializeOwned; const UPLOAD_COMMAND_ID: u32; } @@ -183,8 +195,12 @@ impl Command for UploadTransactionPayload { pub(crate) fn consume( context: &CommandContext<'_, '_>, - reference: TransactionPayloadRef, + input: TransactionCommandInput, ) -> Result { + let reference = match input { + TransactionCommandInput::Inline(input) => return Ok(input), + TransactionCommandInput::Reference(reference) => reference, + }; reference.validate(context.now_ms())?; let mut bytes = Vec::with_capacity(reference.bytes as usize); for chunk in 0..(reference.bytes as usize).div_ceil(CHUNK_BYTES) { diff --git a/tests/elastic_cells.rs b/tests/elastic_cells.rs index 5cfa44d..44f5d10 100644 --- a/tests/elastic_cells.rs +++ b/tests/elastic_cells.rs @@ -99,12 +99,12 @@ use beyonddb::{ ReadPartitionState, ReadPartitionStreamJournal, ReadPartitionTransaction, ReadPendingCrossCellTransactions, ReadPendingCrossCellTransactionsInput, ReadTransactionInput, ReadTtlSchedule, ReadTtlSweep, ReadUnresolvedCoordinatorParticipants, RecordParticipantPrepare, - RecordParticipantResolution, ResolvePartitionTransaction, ResolveTransactionInput, - ResolveTransactionOutcome, RoutePageInput, RoutePageOutcome, SealPartition, - SealPartitionOutcome, SplitPlan, StreamConfig, StreamJournalInput, StreamJournalOutcome, - TableRoute, TableSpec, TransactionOperation, TransactionToken, UpdateTtl, UpdateTtlInput, - account_target, build_http_state, coordinator_target, credential_target, data_key_hash, - data_target, initialize_account, initialize_coordinator, initialize_partition, + RecordParticipantResolution, RecordParticipantResolutions, ResolvePartitionTransaction, + ResolveTransactionInput, ResolveTransactionOutcome, RoutePageInput, RoutePageOutcome, + SealPartition, SealPartitionOutcome, SplitPlan, StreamConfig, StreamJournalInput, + StreamJournalOutcome, TableRoute, TableSpec, TransactionOperation, TransactionToken, UpdateTtl, + UpdateTtlInput, account_target, build_http_state, coordinator_target, credential_target, + data_key_hash, data_target, initialize_account, initialize_coordinator, initialize_partition, }; use cellule_app::CellApplication; use cellule_host::CellNodeBuilder; diff --git a/tests/elastic_cells/account_participant.rs b/tests/elastic_cells/account_participant.rs index 2df2d4b..2f611d7 100644 --- a/tests/elastic_cells/account_participant.rs +++ b/tests/elastic_cells/account_participant.rs @@ -43,7 +43,7 @@ async fn mixed_participants_preserve_locks_and_finish_after_owner_restart() { *account.application().as_bytes(), ); let host = CellNodeBuilder::new(application.clone()) - .with_runtime(SqlWorkerPool::new(1, 16).unwrap(), 16 * 1024 * 1024) + .with_runtime(SqlWorkerPool::new(1, 16).unwrap(), 64 * 1024 * 1024) .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(1 << 30))) .with_session(session) .build_unleased_for_maintenance() @@ -389,7 +389,7 @@ async fn mixed_participants_preserve_locks_and_finish_after_owner_restart() { let next_session = SessionId::from_bytes([214; 16]); let restored = CellNodeBuilder::new(application.clone()) - .with_runtime(SqlWorkerPool::new(1, 16).unwrap(), 16 * 1024 * 1024) + .with_runtime(SqlWorkerPool::new(1, 16).unwrap(), 64 * 1024 * 1024) .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(1 << 30))) .with_session(next_session) .build_unleased_for_maintenance() diff --git a/tests/elastic_cells/public_transactions.rs b/tests/elastic_cells/public_transactions.rs index dc681d2..2792788 100644 --- a/tests/elastic_cells/public_transactions.rs +++ b/tests/elastic_cells/public_transactions.rs @@ -13,7 +13,7 @@ pub(super) async fn assert_lost_replies_and_canceled_token_reuse( ) { let session = SessionId::from_bytes([221; 16]); let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 8).unwrap(), 16 * 1024 * 1024, session).unwrap(); + CellRuntime::new(SqlWorkerPool::new(1, 8).unwrap(), 64 * 1024 * 1024, session).unwrap(); let signer = PeerSigner::new( session, registry.release_digest(), @@ -80,8 +80,8 @@ pub(super) async fn assert_lost_replies_and_canceled_token_reuse( .unwrap(); assert_eq!( lost.load(Ordering::SeqCst), - 127, - "uploads, BEGIN, prepare, decision and resolution replies were dropped after dispatch" + 15, + "inline BEGIN, prepare, decision and resolution replies were dropped after dispatch" ); for info in &infos { assert_eq!( @@ -181,7 +181,10 @@ pub(super) async fn assert_lost_replies_and_canceled_token_reuse( read, vec![Some(item.clone()), Some(new_item), None, Some(item.clone())] ); - assert_eq!(lost.load(Ordering::SeqCst), 127); + assert_eq!(lost.load(Ordering::SeqCst), 15); + // The direct phase-command fixtures below exercise snapshot recovery, + // not lost replies. Keep the injector from consuming their uploads. + lost.store(127, Ordering::SeqCst); super::transaction_reads::assert_shared_snapshots( &client, &storage, diff --git a/tests/elastic_cells/transaction_driver.rs b/tests/elastic_cells/transaction_driver.rs index 185e566..37031b7 100644 --- a/tests/elastic_cells/transaction_driver.rs +++ b/tests/elastic_cells/transaction_driver.rs @@ -289,7 +289,8 @@ async fn driver_resumes_prepares_and_resolves_commit_condition_and_lock_failures } else { Arc::new(RefusePhase { inner: transport, - command: if scenario == 186 { 12 } else { 14 }, + command: 12, + persistent: scenario == 187, winner: (scenario == 186).then(|| (client.clone(), transaction_id)), refused: lost.clone(), }) @@ -313,16 +314,15 @@ async fn driver_resumes_prepares_and_resolves_commit_condition_and_lock_failures .resume_cross_cell_transaction(account_id, &transaction_id, transaction_id) .await .unwrap(); - assert_eq!( - lost.load(Ordering::SeqCst), - if scenario == 185 { - 3 - } else if scenario == 188 { - 2 - } else { - 1 - } - ); + if scenario == 188 { + assert!(lost.load(Ordering::SeqCst) >= 2); + } else { + assert_eq!( + lost.load(Ordering::SeqCst), + if scenario == 185 { 3 } else { 1 }, + "scenario={scenario}" + ); + } peer_runtime.shutdown().await.unwrap(); decision } else { @@ -371,9 +371,10 @@ async fn driver_resumes_prepares_and_resolves_commit_condition_and_lock_failures .await .unwrap(); if let Some(sequence) = recorded_sequence { - // Only the remaining prepare, decision, and two resolutions need - // coordinator commits; a durable prepare must not be recorded twice. - assert_eq!(status.receipt.commit_sequence - sequence, 4); + // The remaining prepare and decision need two commits. Nearby + // resolutions can share one commit; the progress timer may split + // them when a participant finishes later. + assert!((3..=4).contains(&(status.receipt.commit_sequence - sequence))); } let status = status.output.0.unwrap(); assert_eq!(status.resolved_count, 2); @@ -605,6 +606,7 @@ impl PeerRoundTrip for DropPhaseReplies { struct RefusePhase { inner: DropPhaseReplies, command: u32, + persistent: bool, winner: Option<(CellClient, [u8; 16])>, refused: Arc, } @@ -622,6 +624,7 @@ impl PeerRoundTrip for RefusePhase { let dispatcher = self.inner.dispatcher.clone(); let winner = self.winner.clone(); let command = self.command; + let persistent = self.persistent; let refused = self.refused.clone(); Box::pin(async move { use cellule_runtime::peer::wire::{mutation_request, peer_request}; @@ -637,7 +640,8 @@ impl PeerRoundTrip for RefusePhase { let selected = matches!(verified.operation(), Some(peer_request::Operation::Mutate(mutation)) if matches!(&mutation.operation, Some(mutation_request::Operation::CellCommand(input)) if input.command_id == command)); - if selected && refused.swap(1, Ordering::SeqCst) == 0 { + if selected && (persistent || refused.swap(1, Ordering::SeqCst) == 0) { + refused.store(1, Ordering::SeqCst); if let Some((client, transaction_id)) = winner { let storage = CellStorage::new(client, "us-east-1"); assert_eq!( diff --git a/tests/elastic_cells/transaction_resolution.rs b/tests/elastic_cells/transaction_resolution.rs index 11fd782..b0083ce 100644 --- a/tests/elastic_cells/transaction_resolution.rs +++ b/tests/elastic_cells/transaction_resolution.rs @@ -290,31 +290,33 @@ async fn resolution_progress(participant_count: usize, hold_first: bool) { .expect("selected participant RPC must start"); // Query raw coordinator progress: a public item read would help // recovery and conceal a resolver that still waits sequentially. - tokio::time::timeout(Duration::from_secs(10), async { - loop { - let status = client - .query::( - &coordinator, - None, - Json(ReadCrossCellTransactionInput { - account_id: account_id.into(), - transaction_id: id, - routing_key: token.as_bytes().to_vec(), - }), - ) - .await - .unwrap() - .output - .0 - .unwrap(); - if usize::from(status.resolved_count) == participant_count - 1 { - break; + if cut != 3 { + tokio::time::timeout(Duration::from_secs(10), async { + loop { + let status = client + .query::( + &coordinator, + None, + Json(ReadCrossCellTransactionInput { + account_id: account_id.into(), + transaction_id: id, + routing_key: token.as_bytes().to_vec(), + }), + ) + .await + .unwrap() + .output + .0 + .unwrap(); + if usize::from(status.resolved_count) == participant_count - 1 { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; } - tokio::time::sleep(Duration::from_millis(10)).await; - } - }) - .await - .expect("healthy participants must resolve before the stalled RPC returns"); + }) + .await + .expect("healthy participants must resolve before the stalled RPC returns"); + } assert!( !resolving.is_finished(), "incomplete resolution cannot succeed" @@ -333,39 +335,53 @@ async fn resolution_progress(participant_count: usize, hold_first: bool) { } else { ParticipantTransactionState::Aborted }; + let mut completed_positions = vec![false; participant_count]; for (position, (target, participant, _)) in participants.iter().enumerate() { - let input = Json(ReadTransactionInput { - transaction_id: id, - coordinator_cell: *coordinator.cell_id().as_bytes(), - }); - let state = match participant { - CoordinatorParticipantTarget::Account => { - client - .query::(target, None, input) - .await - .unwrap() - .output - .0 - } - CoordinatorParticipantTarget::Data { .. } => { - client - .query::(target, None, input) - .await - .unwrap() - .output - .0 + let read_state = || async { + let input = Json(ReadTransactionInput { + transaction_id: id, + coordinator_cell: *coordinator.cell_id().as_bytes(), + }); + match participant { + CoordinatorParticipantTarget::Account => { + client + .query::(target, None, input) + .await + .unwrap() + .output + .0 + } + CoordinatorParticipantTarget::Data { .. } => { + client + .query::(target, None, input) + .await + .unwrap() + .output + .0 + } } }; - assert_eq!( - state, - if position == 0 && cut != 3 { - ParticipantTransactionState::Prepared - } else { - terminal.clone() - }, - "healthy later participant must finish: commit={commit}, cut={cut}, position={position}" - ); + let state = read_state().await; + completed_positions[position] = state == terminal; + if cut == 3 { + assert!( + state == terminal || state == ParticipantTransactionState::Prepared, + "participant has an unexpected state: commit={commit}, position={position}" + ); + } else { + assert_eq!( + state, + if position == 0 { + ParticipantTransactionState::Prepared + } else { + terminal.clone() + }, + "healthy later participant must finish: commit={commit}, cut={cut}, position={position}" + ); + } } + let healthy_position = if cut == 3 { 0 } else { 1 }; + assert!(completed_positions[healthy_position]); let status = client .query::( &coordinator, @@ -381,11 +397,13 @@ async fn resolution_progress(participant_count: usize, hold_first: bool) { .output .0 .unwrap(); - assert_eq!( - (status.decision, status.resolved_count), - (decision.clone(), (participant_count - 1) as u8) - ); - let healthy_info = &participants[1].2; + assert_eq!(status.decision, decision); + if cut == 3 { + assert!(usize::from(status.resolved_count) < participant_count); + } else { + assert_eq!(usize::from(status.resolved_count), participant_count - 1); + } + let healthy_info = &participants[healthy_position].2; assert_eq!( local.get_item(healthy_info, &key).await.unwrap(), commit.then(|| proposed.clone()) @@ -395,7 +413,7 @@ async fn resolution_progress(participant_count: usize, hold_first: bool) { let mut newer = key.clone(); newer.insert("value".into(), AttributeValue::S("newer".into())); for (position, (_, _, info)) in participants.iter().enumerate() { - if position == 0 && cut != 3 { + if !completed_positions[position] { continue; } // Include the first apply whose coordinator receipt was lost. @@ -456,7 +474,7 @@ async fn resolution_progress(participant_count: usize, hold_first: bool) { for (position, (_, _, info)) in participants.iter().enumerate() { assert_eq!( local.get_item(info, &key).await.unwrap(), - if position > 0 || cut == 3 { + if completed_positions[position] { Some(newer.clone()) } else { commit.then(|| proposed.clone()) @@ -529,14 +547,16 @@ impl PeerRoundTrip for ResolutionFault { ) } 3 => command.is_some_and(|command| { - if command.command_id != RecordParticipantResolution::ID { + if command.command_id != RecordParticipantResolutions::ID { return false; } let mut decoder = BoundedDecoder::new(&command.input, 4096).unwrap(); - let phase = Json::::decode(&mut decoder) + let phases = Json::>::decode(&mut decoder) .unwrap() .0; - phase.participant_cell == *first.cell_id().as_bytes() + phases + .iter() + .any(|phase| phase.participant_cell == *first.cell_id().as_bytes()) }), _ => false, }; diff --git a/tests/elastic_cells/transaction_transport.rs b/tests/elastic_cells/transaction_transport.rs index de56ae3..ceee757 100644 --- a/tests/elastic_cells/transaction_transport.rs +++ b/tests/elastic_cells/transaction_transport.rs @@ -1,5 +1,5 @@ use crate::*; -use beyonddb::{TransactionPayloadRef, UploadTransactionPayload}; +use beyonddb::{TransactionCommandInput, TransactionPayloadRef, UploadTransactionPayload}; fn mutation() -> MutationIdentity { let identity = identity(246); @@ -115,7 +115,11 @@ async fn upload_seals_only_complete_immutable_inputs_after_owner_replacement() { ); assert!( client - .command::(&coordinator, mutation(), Json(reference.clone())) + .command::( + &coordinator, + mutation(), + Json(TransactionCommandInput::Reference(reference.clone())), + ) .await .is_err() ); @@ -156,7 +160,11 @@ async fn upload_seals_only_complete_immutable_inputs_after_owner_replacement() { .unwrap(); assert!( client - .command::(&coordinator, mutation(), Json(forged)) + .command::( + &coordinator, + mutation(), + Json(TransactionCommandInput::Reference(forged)), + ) .await .is_err() ); @@ -235,18 +243,30 @@ async fn upload_seals_only_complete_immutable_inputs_after_owner_replacement() { } let identity = mutation(); let first = client - .command::(&coordinator, identity, Json(reference.clone())) + .command::( + &coordinator, + identity, + Json(TransactionCommandInput::Reference(reference.clone())), + ) .await .unwrap(); assert_eq!(first.output.0, BeginCrossCellTransactionOutcome::Begun); let replay = client - .command::(&coordinator, identity, Json(reference.clone())) + .command::( + &coordinator, + identity, + Json(TransactionCommandInput::Reference(reference.clone())), + ) .await .unwrap(); assert_eq!(first.receipt, replay.receipt); assert_eq!( client - .command::(&coordinator, mutation(), Json(competing)) + .command::( + &coordinator, + mutation(), + Json(TransactionCommandInput::Reference(competing)), + ) .await .unwrap() .output @@ -260,7 +280,11 @@ async fn upload_seals_only_complete_immutable_inputs_after_owner_replacement() { // payload remains authoritative and is independently readable in pieces. assert!( client - .command::(&coordinator, mutation(), Json(reference)) + .command::( + &coordinator, + mutation(), + Json(TransactionCommandInput::Reference(reference)), + ) .await .is_err() ); @@ -516,7 +540,9 @@ async fn abort_during_upload_fences_delayed_account_and_data_prepares() { .command::( target, mutation(), - Json(references[index].clone()), + Json(TransactionCommandInput::Reference( + references[index].clone(), + )), ) .await } else { @@ -524,7 +550,9 @@ async fn abort_during_upload_fences_delayed_account_and_data_prepares() { .command::( target, mutation(), - Json(references[index].clone()), + Json(TransactionCommandInput::Reference( + references[index].clone(), + )), ) .await }; diff --git a/tests/node_log_authority.rs b/tests/node_log_authority.rs new file mode 100644 index 0000000..0d1afa7 --- /dev/null +++ b/tests/node_log_authority.rs @@ -0,0 +1,195 @@ +use std::{ + sync::Arc, + time::{Duration, SystemTime, UNIX_EPOCH}, +}; + +use beyonddb::NodeLeasePublisher; +use cellule_runtime::{ + identity::{Digest, NodeId, SessionId}, + ltx::CellStorageLayout, + node::{ + NODE_LOG_PROTOCOL_VERSION, NodeAdvertisement, NodeCapacity, NodeDirectory, + NodeFailureDomain, durability::NodeLogAuthority, log::DurabilityGate, + }, +}; +use cellule_store::Store; +use ed25519_dalek::SigningKey; +use object_store::{memory::InMemory, path::Path}; +use tokio_util::sync::CancellationToken; + +const FLEET: Digest = Digest::from_bytes([80; 32]); +const IMAGE: Digest = Digest::from_bytes([81; 32]); +const RELEASE: Digest = Digest::from_bytes([82; 32]); + +fn now_ms() -> i64 { + i64::try_from( + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_millis(), + ) + .unwrap() +} + +fn advertisement( + node: NodeId, + session: SessionId, + key: u8, + capacity: NodeCapacity, + now: i64, + expires: i64, +) -> cellule_runtime::Result { + NodeAdvertisement::sign( + node, + session, + format!("https://node-{key}.internal:8081"), + FLEET, + Digest::from_bytes([key; 32]), + IMAGE, + RELEASE, + &SigningKey::from_bytes(&[key; 32]), + 1, + now, + expires, + vec![Digest::from_bytes([86; 32])], + vec![1], + NodeFailureDomain::default(), + capacity, + ) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn published_log_authority_reconciles_heartbeat_and_fences_transitions() { + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("beyonddb-node-log-authority"), + [42; 16], + ); + let directory = NodeDirectory::new(layout, FLEET, IMAGE, RELEASE); + let leader_node = NodeId::from_bytes([83; 16]); + let leader_session = SessionId::from_bytes([90; 16]); + let follower_node = NodeId::from_bytes([92; 16]); + let follower_session = SessionId::from_bytes([93; 16]); + let now = now_ms(); + directory + .create( + advertisement( + follower_node, + follower_session, + 95, + NodeCapacity { + free_memory_bytes: 16 * 1024 * 1024, + free_disk_bytes: 1 << 30, + follower_free_bytes: 1 << 30, + job_credits: 8, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + ..NodeCapacity::default() + }, + now, + now + 15_000, + ) + .unwrap(), + now, + ) + .await + .unwrap(); + let published = NodeLeasePublisher::new(directory.clone(), move |now, expires| { + advertisement( + leader_node, + leader_session, + 85, + NodeCapacity { + free_memory_bytes: 16 * 1024 * 1024, + free_disk_bytes: 1 << 30, + job_credits: 8, + ..NodeCapacity::default() + }, + now, + expires, + ) + }) + .publish() + .await + .unwrap(); + let guard = published.guard(); + let authority = published.log_authority(); + assert_eq!( + authority.recruit(1, 4096, 16).await.unwrap(), + Some(vec![follower_node]) + ); + assert_eq!( + authority.recruit(1, 4096, 16).await.unwrap(), + Some(vec![follower_node]) + ); + assert!(authority.activate(2).await.is_err()); + + let cancellation = CancellationToken::new(); + let run_cancellation = cancellation.clone(); + let task = tokio::spawn(async move { published.run(&run_cancellation).await }); + tokio::time::timeout(Duration::from_secs(8), async { + loop { + let observed = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + if observed.advertisement().generation() > 2 { + assert_eq!(observed.advertisement().log().unwrap().epoch(), 1); + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + + authority.activate(1).await.unwrap(); + authority.activate(1).await.unwrap(); + assert!(authority.recruit(1, 4096, 16).await.is_err()); + let gate = DurabilityGate::new(leader_session, leader_node, 1, [follower_node]).unwrap(); + let ticket = gate.issue(1).unwrap(); + assert_eq!(gate.prove_object(ticket).unwrap(), 1); + authority.advance_coverage(1, 1).await.unwrap(); + authority.advance_coverage(1, 1).await.unwrap(); + let observed = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + let log = observed.advertisement().log().unwrap(); + assert!(log.active()); + assert_eq!(log.tiered_through(), 1); + + let barrier = gate.begin_rotation().unwrap(); + authority.close(&barrier).await.unwrap(); + let observed = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + assert!(observed.advertisement().log().is_none()); + let closed_generation = observed.advertisement().generation(); + tokio::time::timeout(Duration::from_secs(8), async { + loop { + let refreshed = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + if refreshed.advertisement().generation() > closed_generation { + assert!(refreshed.advertisement().log().is_none()); + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + cancellation.cancel(); + task.await.unwrap().unwrap(); + guard.fence(); + assert!(matches!( + authority.activate(1).await, + Err(cellule_runtime::Error::Fenced) + )); +} diff --git a/tests/server_binary.rs b/tests/server_binary.rs index 875a107..e9a6e8d 100644 --- a/tests/server_binary.rs +++ b/tests/server_binary.rs @@ -6,7 +6,9 @@ mod server_binary { mod capacity; mod coordinator_recovery; pub(super) mod global_indexes; + mod large_reads; pub(super) mod local_indexes; + mod metadata_cache; } use std::{ @@ -281,6 +283,13 @@ struct ProcessFixture { } async fn process_fixture(initial_partitions: u32) -> ProcessFixture { + process_fixture_with_cache(initial_partitions, false).await +} + +async fn process_fixture_with_cache( + initial_partitions: u32, + auth_cache_enabled: bool, +) -> ProcessFixture { let root = tempfile::tempdir().unwrap(); tls_files(root.path()); let s3 = free_addr(); @@ -358,6 +367,7 @@ async fn process_fixture(initial_partitions: u32) -> ProcessFixture { "owned_accounts": ["123456789012"], "owned_access_keys": ["AKIAIOSFODNN7EXAMPLE"], "initial_partitions": initial_partitions, + "auth_cache_enabled": auth_cache_enabled, "bootstrap": { "account_id": "123456789012", "access_key_id": "AKIAIOSFODNN7EXAMPLE", diff --git a/tests/server_binary/large_reads.rs b/tests/server_binary/large_reads.rs new file mode 100644 index 0000000..9d50322 --- /dev/null +++ b/tests/server_binary/large_reads.rs @@ -0,0 +1,79 @@ +use std::collections::HashMap; + +use aws_sdk_dynamodb::types::{ + AttributeDefinition, AttributeValue, BillingMode, Get, KeySchemaElement, KeyType, + ScalarAttributeType, TransactGetItem, +}; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +#[ignore = "requires Docker, aws CLI, and the pinned RustFS GA image"] +async fn oversized_single_cell_transaction_read_uses_saved_images() { + let mut fixture = crate::process_fixture_with_cache(1, true).await; + let table = "ProcessLargeRead"; + fixture + .sdk + .create_table() + .table_name(table) + .billing_mode(BillingMode::PayPerRequest) + .key_schema( + KeySchemaElement::builder() + .attribute_name("id") + .key_type(KeyType::Hash) + .build() + .unwrap(), + ) + .attribute_definitions( + AttributeDefinition::builder() + .attribute_name("id") + .attribute_type(ScalarAttributeType::S) + .build() + .unwrap(), + ) + .send() + .await + .unwrap(); + let mut reads = Vec::new(); + let mut expected = Vec::new(); + for index in 0..10 { + let key = HashMap::from([("id".into(), AttributeValue::S(format!("key-{index}")))]); + let mut item = key.clone(); + item.insert( + "payload".into(), + AttributeValue::B(vec![0xa5; 380 * 1024].into()), + ); + fixture + .sdk + .put_item() + .table_name(table) + .set_item(Some(item.clone())) + .send() + .await + .unwrap(); + reads.push( + TransactGetItem::builder() + .get( + Get::builder() + .table_name(table) + .set_key(Some(key)) + .build() + .unwrap(), + ) + .build(), + ); + expected.push(item); + } + reads.reverse(); + expected.reverse(); + let result = fixture + .sdk + .transact_get_items() + .set_transact_items(Some(reads)) + .send() + .await + .unwrap(); + assert_eq!(result.responses().len(), expected.len()); + for (response, item) in result.responses().iter().zip(expected) { + assert_eq!(response.item(), Some(&item)); + } + crate::stop(&mut fixture.child, &fixture.log); +} diff --git a/tests/server_binary/local_indexes.rs b/tests/server_binary/local_indexes.rs index 3ac216e..e95208d 100644 --- a/tests/server_binary/local_indexes.rs +++ b/tests/server_binary/local_indexes.rs @@ -12,6 +12,14 @@ use aws_sdk_dynamodb::{ const TABLE: &str = "ProcessLocalIndex"; const INDEX: &str = "ByScore"; +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +#[ignore = "requires Docker, aws CLI, and the pinned RustFS GA image"] +async fn segmented_local_index_scan_advances_empty_pages() { + let mut fixture = crate::process_fixture(4).await; + create(&fixture.sdk).await; + crate::stop(&mut fixture.child, &fixture.log); +} + fn key(pk: &str, sk: &str) -> HashMap { HashMap::from([ ("pk".into(), AttributeValue::S(pk.into())), diff --git a/tests/server_binary/metadata_cache.rs b/tests/server_binary/metadata_cache.rs new file mode 100644 index 0000000..c003ec6 --- /dev/null +++ b/tests/server_binary/metadata_cache.rs @@ -0,0 +1,75 @@ +use crate::*; + +use aws_sdk_dynamodb::types::{ + AttributeDefinition, BillingMode, KeySchemaElement, KeyType, ScalarAttributeType, +}; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +#[ignore = "requires Docker, aws CLI, and the pinned RustFS GA image"] +async fn cached_metadata_observes_local_create_and_update() { + let mut fixture = process_fixture_with_cache(1, true).await; + let before = fixture.sdk.list_tables().send().await.unwrap(); + assert!(!before.table_names().contains(&"ProcessData".to_owned())); + + fixture + .sdk + .create_table() + .table_name("ProcessData") + .key_schema( + KeySchemaElement::builder() + .attribute_name("pk") + .key_type(KeyType::Hash) + .build() + .unwrap(), + ) + .attribute_definitions( + AttributeDefinition::builder() + .attribute_name("pk") + .attribute_type(ScalarAttributeType::S) + .build() + .unwrap(), + ) + .billing_mode(BillingMode::PayPerRequest) + .send() + .await + .unwrap(); + + let after = fixture.sdk.list_tables().send().await.unwrap(); + assert!(after.table_names().contains(&"ProcessData".to_owned())); + let before_update = fixture + .sdk + .describe_table() + .table_name("ProcessData") + .send() + .await + .unwrap(); + assert_eq!( + before_update + .table() + .and_then(|table| table.deletion_protection_enabled()), + Some(false), + ); + + fixture + .sdk + .update_table() + .table_name("ProcessData") + .deletion_protection_enabled(true) + .send() + .await + .unwrap(); + let after_update = fixture + .sdk + .describe_table() + .table_name("ProcessData") + .send() + .await + .unwrap(); + assert_eq!( + after_update + .table() + .and_then(|table| table.deletion_protection_enabled()), + Some(true), + ); + stop(&mut fixture.child, &fixture.log); +} diff --git a/tests/support/transaction_transport.rs b/tests/support/transaction_transport.rs index 0421016..d1fe6a1 100644 --- a/tests/support/transaction_transport.rs +++ b/tests/support/transaction_transport.rs @@ -26,7 +26,11 @@ macro_rules! transaction_command { .unwrap(); } client - .command::<$command>(target, identity, beyonddb::Json(reference)) + .command::<$command>( + target, + identity, + beyonddb::Json(beyonddb::TransactionCommandInput::Reference(reference)), + ) .await } };