From a1c48fcd009b214c442bcf75c6b80126ec0bea2f Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 6 Oct 2026 20:06:20 -0700 Subject: [PATCH 1/8] Pack small Cell root dependencies and verify publication costs Add bounded packed LTX/index objects and inline directory leaves across publication, restore, sparse reads, compaction, backup and collection. Share one verified packed compaction fetch between body and index spools. Add response/proof/worker/peer/storage telemetry and pinned Docker workload, warm/cold exact-retry audits, failure-preserving reports and explicit delivery gaps. Keep raw evidence outside Git. --- .gitignore | 2 + crates/cellule-axum/examples/capacity/mod.rs | 153 +++++- .../cellule-axum/examples/fleet/authority.rs | 3 +- crates/cellule-axum/examples/fleet/mod.rs | 9 +- crates/cellule-axum/examples/fleet/server.rs | 47 +- .../cellule-axum/examples/fleet/transport.rs | 30 +- crates/cellule-axum/examples/sql.rs | 66 ++- .../examples/sql_metrics/capture.rs | 8 + .../cellule-axum/examples/sql_metrics/mod.rs | 222 +++++++- .../examples/sql_metrics/storage.rs | 166 ++++++ .../examples/sql_metrics/tests.rs | 174 +++++++ .../writer_tests/successors/mod.rs | 2 +- crates/cellule-ltx/docs/README.md | 16 +- crates/cellule-ltx/docs/packed-root-format.md | 73 +++ crates/cellule-ltx/src/cell_layout.rs | 3 + .../cellule-ltx/src/replica/compaction/mod.rs | 74 ++- .../src/replica/compaction/source.rs | 348 ++++++++----- .../src/replica/directory/initial.rs | 13 +- .../cellule-ltx/src/replica/directory/mod.rs | 21 +- .../src/replica/directory/relocate.rs | 1 + .../src/replica/directory/tests.rs | 2 + crates/cellule-ltx/src/replica/mod.rs | 8 + crates/cellule-ltx/src/replica/packed.rs | 139 +++++ crates/cellule-ltx/src/replica/prepare.rs | 58 ++- crates/cellule-ltx/src/replica/root.rs | 89 +++- crates/cellule-ltx/src/replica/upload.rs | 18 +- crates/cellule-ltx/src/replica/verify.rs | 29 ++ crates/cellule-ltx/tests/cell/restore.rs | 2 +- crates/cellule-ltx/tests/cell/roots.rs | 1 + .../tests/cell/roots/coalescing.rs | 4 +- .../tests/cell/roots/compaction.rs | 98 +++- .../tests/cell/roots/compaction_transfers.rs | 21 +- .../cellule-ltx/tests/cell/roots/lifecycle.rs | 18 +- crates/cellule-ltx/tests/cell/roots/packed.rs | 145 ++++++ .../tests/cell/roots/prepare_cost.rs | 12 +- .../tests/host/hooks/compaction.rs | 53 +- .../cellule-ltx/tests/host/hooks/injection.rs | 7 +- .../cellule-ltx/tests/host/hooks/prepare.rs | 26 +- .../src/performance_tests.rs | 2 + crates/cellule-runtime/api-prelude.txt | 1 + .../src/cell/actor/inventory/tests.rs | 1 + crates/cellule-runtime/src/cell/actor/mod.rs | 15 + .../src/cell/actor/requests.rs | 11 + .../cellule-runtime/src/cell/actor/runtime.rs | 12 + .../cellule-runtime/src/cell/actor/state.rs | 4 + crates/cellule-runtime/src/cell/actor/task.rs | 32 +- .../src/cell/actor/tasks/activation.rs | 1 + .../src/cell/actor/tasks/publication.rs | 6 + crates/cellule-runtime/src/fleet/telemetry.rs | 29 ++ crates/cellule-runtime/src/follower/mod.rs | 38 ++ .../src/follower/records/append.rs | 17 +- .../src/follower/tests/append.rs | 60 +++ crates/cellule-runtime/src/lib.rs | 2 +- .../src/node/durability/mod.rs | 5 + crates/cellule-runtime/src/node/log/mod.rs | 44 ++ crates/cellule-runtime/src/node/log/tests.rs | 22 + .../cellule-runtime/src/publication/tests.rs | 4 +- .../src/recovery/retention/mod.rs | 5 +- .../cellule-runtime/tests/runtime/backup.rs | 2 +- .../durability/admission_batching.rs | 8 + .../runtime/lifecycle/durability/group.rs | 18 + .../runtime/lifecycle/ownership/prefix.rs | 2 +- crates/cellule-store/src/observation/mod.rs | 86 +++- crates/cellule-store/src/observation/tests.rs | 50 ++ docs/write-performance-delivery.md | 115 +++++ docs/write-performance-proposal.md | 3 + scripts/perf/README.md | 113 ++++ scripts/perf/build.py | 157 ++++++ scripts/perf/celld/index.js | 58 +++ scripts/perf/celld/wrangler.json | 25 + scripts/perf/compare.py | 97 ++++ scripts/perf/control.py | 42 ++ scripts/perf/http_audit.rs | 132 +++++ scripts/perf/report.py | 225 ++++++++ scripts/perf/run.py | 485 ++++++++++++++++++ scripts/perf/wait-store-ready.py | 35 ++ scripts/tests/test_perf_report.py | 135 +++++ 77 files changed, 3938 insertions(+), 322 deletions(-) create mode 100644 crates/cellule-axum/examples/sql_metrics/storage.rs create mode 100644 crates/cellule-ltx/docs/packed-root-format.md create mode 100644 crates/cellule-ltx/src/replica/packed.rs create mode 100644 crates/cellule-ltx/tests/cell/roots/packed.rs create mode 100644 docs/write-performance-delivery.md create mode 100644 scripts/perf/README.md create mode 100644 scripts/perf/build.py create mode 100644 scripts/perf/celld/index.js create mode 100644 scripts/perf/celld/wrangler.json create mode 100644 scripts/perf/compare.py create mode 100644 scripts/perf/control.py create mode 100644 scripts/perf/http_audit.rs create mode 100644 scripts/perf/report.py create mode 100644 scripts/perf/run.py create mode 100644 scripts/perf/wait-store-ready.py create mode 100644 scripts/tests/test_perf_report.py diff --git a/.gitignore b/.gitignore index 332bf20d..d89a3afd 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,5 @@ /target .DS_Store node_modules/ +__pycache__/ +*.pyc diff --git a/crates/cellule-axum/examples/capacity/mod.rs b/crates/cellule-axum/examples/capacity/mod.rs index c04a6439..bb62d123 100644 --- a/crates/cellule-axum/examples/capacity/mod.rs +++ b/crates/cellule-axum/examples/capacity/mod.rs @@ -27,6 +27,13 @@ pub struct Config { warmup_seconds: u64, seconds: u64, evidence_directory: std::path::PathBuf, + seed_file: Option, + #[serde(default)] + write_offset: u64, + #[serde(default)] + metrics_urls: Vec, + hot_read_cells: Option, + metrics_tls_directory: Option, } impl Config { @@ -40,6 +47,17 @@ impl Config { || self.write_rate + self.read_rate == 0 || !(1..=60).contains(&self.warmup_seconds) || !(1..=3_600).contains(&self.seconds) + || !self.write_offset.is_multiple_of(self.cells as u64) + || self + .hot_read_cells + .is_some_and(|cells| cells == 0 || cells > self.cells) + || self.metrics_urls.len() > 3 + || self.metrics_urls.iter().any(|url| !valid_metrics_url(url)) + || (self + .metrics_urls + .iter() + .any(|url| url.starts_with("https:")) + && self.metrics_tls_directory.is_none()) { return Err("invalid bounded loopback capacity configuration".into()); } @@ -47,6 +65,44 @@ impl Config { } } +fn valid_metrics_url(value: &str) -> bool { + reqwest::Url::parse(value).is_ok_and(|url| { + matches!(url.scheme(), "http" | "https") + && url.host_str().is_some_and(|host| { + host.parse::() + .is_ok_and(|ip| ip.is_loopback()) + }) + && url.username().is_empty() + && url.password().is_none() + }) +} + +async fn sample_metrics( + client: &reqwest::Client, + urls: &[String], + destination: &std::path::Path, +) -> Result<()> { + let mut observations = Vec::new(); + for url in urls { + let started = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH)? + .as_millis(); + let result: serde_json::Value = client + .get(url) + .send() + .await? + .error_for_status()? + .json() + .await?; + let finished = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH)? + .as_millis(); + observations.push(serde_json::json!({ "url": url, "request_started_ms": started, "request_finished_ms": finished, "metrics": result })); + } + std::fs::write(destination, serde_json::to_vec(&observations)?)?; + Ok(()) +} + #[derive(Clone, Copy)] enum Kind { Write, @@ -97,16 +153,45 @@ pub async fn run(config: Config) -> Result<()> { config.evidence_directory.join("config.json"), serde_json::to_vec_pretty(&config)?, )?; + let mut metrics_client = reqwest::Client::builder() + .http1_only() + .no_proxy() + .timeout(Duration::from_secs(5)); + if let Some(path) = &config.metrics_tls_directory { + metrics_client = metrics_client.add_root_certificate(reqwest::Certificate::from_pem( + &std::fs::read(path.join("ca.crt"))?, + )?); + let mut identity = std::fs::read(path.join("node-0.crt"))?; + identity.extend(std::fs::read(path.join("node-0.key"))?); + metrics_client = metrics_client.identity(reqwest::Identity::from_pem(&identity)?); + } + let metrics_client = metrics_client.build()?; let client = reqwest::Client::builder() .http1_only() .no_proxy() .timeout(Duration::from_secs(60)) .build()?; let url = format!("http://{}", config.address); - let mut seeds = Vec::with_capacity(config.cells); + let mut seeds: Vec = Vec::with_capacity(config.cells); let mut setup = journal(&config.evidence_directory.join("setup.jsonl"))?; let setup_started = Instant::now(); - for index in 0..config.cells { + if let Some(path) = &config.seed_file { + seeds = serde_json::from_slice(&std::fs::read(path)?)?; + if seeds.len() != config.cells + || seeds.iter().enumerate().any(|(index, seed)| { + seed.output + .as_ref() + .is_none_or(|order| order.id != index as i64) + }) + { + return Err("seed file does not contain one ordered acknowledged row per Cell".into()); + } + } + for index in 0..if config.seed_file.is_none() { + config.cells + } else { + 0 + } { let id = i64::try_from(index)?; let initial: Observation = client .get(format!("{url}/orders/{id}")) @@ -149,9 +234,53 @@ pub async fn run(config: Config) -> Result<()> { setup.get_ref().sync_all()?; let config = Arc::new(config); let seeds = Arc::new(seeds); + if !config.metrics_urls.is_empty() { + // Warm exporter allocations and the dedicated mTLS sampling client + // before scheduled arrivals. Priming is recorded outside the window. + sample_metrics( + &metrics_client, + &config.metrics_urls, + &config.evidence_directory.join("metrics-prime.json"), + ) + .await?; + } let start = Instant::now(); let warm_end = start + Duration::from_secs(config.warmup_seconds); let end = warm_end + Duration::from_secs(config.seconds); + let metrics_task = { + let config = config.clone(); + let client = metrics_client; + tokio::spawn(async move { + if !config.metrics_urls.is_empty() { + tokio::time::sleep_until(warm_end).await; + sample_metrics( + &client, + &config.metrics_urls, + &config.evidence_directory.join("metrics-window-start.json"), + ) + .await?; + for minute in 1..=(config.seconds.saturating_sub(1) / 60) { + tokio::time::sleep_until(warm_end + Duration::from_secs(minute * 60)).await; + sample_metrics( + &client, + &config.metrics_urls, + &config + .evidence_directory + .join(format!("metrics-window-minute-{minute}.json")), + ) + .await?; + } + tokio::time::sleep_until(end).await; + sample_metrics( + &client, + &config.metrics_urls, + &config.evidence_directory.join("metrics-window-end.json"), + ) + .await?; + } + Ok::<_, Box>(()) + }) + }; let (sender, receiver) = mpsc::channel::(config.queue_capacity); let receiver = Arc::new(Mutex::new(receiver)); let mut tasks = JoinSet::new(); @@ -174,11 +303,22 @@ pub async fn run(config: Config) -> Result<()> { loop { let job = receiver.lock().await.recv().await; let Some(job) = job else { break }; - let shard = usize::try_from(job.index % config.cells as u64)?; + let shard = usize::try_from( + job.index + % match job.kind { + Kind::Read => config.hot_read_cells.unwrap_or(config.cells), + Kind::Write => config.cells, + } as u64, + )?; let started = Instant::now(); let (body, read_id, result) = match job.kind { Kind::Write => { - let id = i64::try_from(config.cells as u64 + job.index)?; + let id = i64::try_from( + (config.cells as u64) + .checked_add(job.index) + .and_then(|id| id.checked_add(config.write_offset)) + .ok_or("write ID overflow")?, + )?; let body = WriteRequest::new(id)?; let result = request::post(&client, &url, &body, &seeds[shard].receipt).await; @@ -289,6 +429,11 @@ pub async fn run(config: Config) -> Result<()> { Err(error) => task_errors.push(error.to_string()), } } + match metrics_task.await { + Ok(Ok(())) => {} + Ok(Err(error)) => task_errors.push(format!("measurement sampling failed: {error}")), + Err(error) => task_errors.push(format!("measurement sampling join failed: {error}")), + } let summary = serde_json::json!({ "setup_seconds": setup_started.elapsed().as_secs_f64() - start.elapsed().as_secs_f64(), "window_seconds": config.seconds, "drain_seconds": Instant::now().saturating_duration_since(end).as_secs_f64(), diff --git a/crates/cellule-axum/examples/fleet/authority.rs b/crates/cellule-axum/examples/fleet/authority.rs index eb0b1c0b..b8467239 100644 --- a/crates/cellule-axum/examples/fleet/authority.rs +++ b/crates/cellule-axum/examples/fleet/authority.rs @@ -84,7 +84,8 @@ impl Enrollment { } } .await; - if result.is_err() { + if let Err(error) = &result { + eprintln!("Node heartbeat failed: {error:?}"); running.lease.fence(); } result diff --git a/crates/cellule-axum/examples/fleet/mod.rs b/crates/cellule-axum/examples/fleet/mod.rs index 751dbf27..b0df32f1 100644 --- a/crates/cellule-axum/examples/fleet/mod.rs +++ b/crates/cellule-axum/examples/fleet/mod.rs @@ -1,4 +1,5 @@ //! Optional three-process, real-directory follower durability benchmark wiring. +use super::sql_metrics::{PeerPhase, QueryMetrics}; use cellule_peer_http::LoadedPeerTls; use cellule_runtime::identity::NodeId; use cellule_runtime::node::NodeDirectory; @@ -67,8 +68,9 @@ impl Config { layout: cellule_ltx::CellStorageLayout, code: Digest, bind: std::net::SocketAddr, + metrics: Arc, ) -> Result<()> { - server::serve(self, layout, code, bind).await + server::serve(self, layout, code, bind, metrics).await } pub async fn start_owner( @@ -78,6 +80,7 @@ impl Config { session: SessionId, application: ApplicationId, runtime: &CellRuntime, + metrics: Arc, ) -> Result { let tls = self.tls()?; let directory = NodeDirectory::new(layout, tls.fleet(), code, code); @@ -119,7 +122,9 @@ impl Config { runtime.install_node_lease(enrollment.authority.lease.clone())?; let peers = enrollment.authority.recruit().await?; let members = peers.iter().map(|peer| peer.node()).collect(); - let transport = Arc::new(transport::Transport::new(directory, session, tls, peers)?); + let transport = Arc::new(transport::Transport::new( + directory, session, tls, peers, metrics, + )?); let config = cellule_runtime::node::durability::NodeDurabilityConfig::new( session, node(0), diff --git a/crates/cellule-axum/examples/fleet/server.rs b/crates/cellule-axum/examples/fleet/server.rs index 42847af1..98121504 100644 --- a/crates/cellule-axum/examples/fleet/server.rs +++ b/crates/cellule-axum/examples/fleet/server.rs @@ -5,7 +5,7 @@ use axum::{ extract::{ConnectInfo, DefaultBodyLimit, State}, http::StatusCode, middleware, - routing::post, + routing::{get, post}, }; use bytes::Bytes; use cellule_peer_http::PeerTlsIdentity; @@ -23,6 +23,7 @@ struct Server { store: FollowerStore, lease: NodeLeaseGuard, jobs: TaskTracker, + metrics: Arc, } pub(super) async fn serve( @@ -30,6 +31,7 @@ pub(super) async fn serve( layout: cellule_ltx::CellStorageLayout, code: Digest, bind: std::net::SocketAddr, + metrics: Arc, ) -> Result<()> { let tls = config.tls()?; let directory = NodeDirectory::new(layout, tls.fleet(), code, code); @@ -38,7 +40,10 @@ pub(super) async fn serve( config.root.join(format!("data-{}", config.index)), cellule_ltx::Limits::default(), cellule_ltx::DiskBudget::new(1 << 30), - )?; + )? + .with_telemetry( + cellule_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(metrics.clone()), + ); let listener = tokio::net::TcpListener::bind(bind) .await .map_err(transport_error)?; @@ -64,13 +69,13 @@ pub(super) async fn serve( store, lease: enrollment.authority.lease.clone(), jobs: jobs.clone(), + metrics: metrics.clone(), }; // Reserve the body/decoding lifetime before Axum buffers request bytes. // The follower has one ordered append lane; overload never allocates another body. let admission = Arc::new(Semaphore::new(1)); let router = Router::new() .route(wire::PATH, post(handle)) - .with_state(state) .layer(DefaultBodyLimit::max(wire::MAX_REQUEST_BYTES)) .layer(middleware::from_fn( move |mut request: axum::extract::Request, next: middleware::Next| { @@ -83,7 +88,14 @@ pub(super) async fn serve( next.run(request).await } }, - )); + )) + .route( + "/debug/metrics", + get(|State(server): State| async move { + axum::Json(server.metrics.window_snapshot()) + }), + ) + .with_state(state); println!( "Follower service: {}; member: {:?}; session: {:?}", endpoint, @@ -101,7 +113,9 @@ pub(super) async fn serve( .await; jobs.close(); jobs.wait().await; + println!("Follower metrics before drain: {}", metrics.snapshot()); let stopped = enrollment.stop().await; + println!("Follower metrics: {}", metrics.snapshot()); served.map_err(transport_error)?; signal_rx .await @@ -136,9 +150,13 @@ async fn handle( async fn handle_inner(server: Server, peer: PeerTlsIdentity, encoded: Bytes) -> Result> { server.lease.check()?; + let phase = std::time::Instant::now(); let key = VerifyingKey::from_bytes(&peer.public_key()).map_err(Error::PeerSignature)?; let body = wire::verify(&encoded, &key, wire::REQUEST_DOMAIN)?; let request = wire::Request::decode(body.as_slice())?; + server + .metrics + .peer_phase(PeerPhase::RequestVerify, phase.elapsed()); let started_at = std::time::Instant::now(); let started_ms = clock()?; validate_request(&request, started_ms, "before directory verification")?; @@ -148,9 +166,15 @@ async fn handle_inner(server: Server, peer: PeerTlsIdentity, encoded: Bytes) -> return Err(Error::PeerAuthorization("capacity log member differs")); } let enrolled = server - .directory - .peer_verifier(sender, peer.certificate(), peer.public_key(), clock()?) - .await?; + .metrics + .enrollment( + true, + server + .directory + .peer_verifier(sender, peer.certificate(), peer.public_key(), clock()?), + ) + .await; + let enrolled = enrolled?; // Directory I/O consumed time. Preserve the accepted horizon across wall // clock rollback, while monotonic elapsed time still expires the request. let now = wire::request_time(started_ms, clock()?, started_at.elapsed())?; @@ -172,7 +196,8 @@ async fn handle_inner(server: Server, peer: PeerTlsIdentity, encoded: Bytes) -> request.covered_through, clock()?, )?; - let receipt = server + let phase = std::time::Instant::now(); + let result = server .store .append( leader, @@ -180,7 +205,11 @@ async fn handle_inner(server: Server, peer: PeerTlsIdentity, encoded: Bytes) -> request.frames.into_iter().map(Bytes::from).collect(), request.covered_through, ) - .await?; + .await; + server + .metrics + .peer_phase(PeerPhase::DurableAppend, phase.elapsed()); + let receipt = result?; reply.base_sequence = receipt.base_sequence; reply.durable_through = receipt.durable_through; } diff --git a/crates/cellule-axum/examples/fleet/transport.rs b/crates/cellule-axum/examples/fleet/transport.rs index 7520407e..a8715bff 100644 --- a/crates/cellule-axum/examples/fleet/transport.rs +++ b/crates/cellule-axum/examples/fleet/transport.rs @@ -20,6 +20,7 @@ pub(super) struct Transport { sender: SessionId, tls: Arc, members: HashMap, + metrics: Arc, } impl Transport { @@ -28,6 +29,7 @@ impl Transport { sender: SessionId, tls: Arc, members: Vec, + metrics: Arc, ) -> Result { let mut enrolled = HashMap::new(); for original in members { @@ -47,6 +49,7 @@ impl Transport { sender, tls, members: enrolled, + metrics, }) } @@ -56,10 +59,14 @@ impl Transport { .get(&member) .ok_or(Error::PeerAuthorization("unknown capacity follower"))?; let fresh = self - .directory - .load_if_live(peer.original.session(), clock()?) - .await? - .ok_or(Error::Fenced)?; + .metrics + .enrollment( + false, + self.directory + .load_if_live(peer.original.session(), clock()?), + ) + .await; + let fresh = fresh?.ok_or(Error::Fenced)?; let actual = fresh.advertisement(); if actual.node() != member || actual.certificate() != peer.original.certificate() @@ -71,13 +78,17 @@ impl Transport { request.sender = self.sender.as_bytes().to_vec(); request.member = member.as_bytes().to_vec(); request.deadline_ms = clock()?.checked_add(10_000).ok_or(Error::Deadline)?; + let phase = std::time::Instant::now(); let body = request.encode_to_vec(); let digest = blake3::hash(&body); let encoded = wire::sign(body, self.tls.signing_key(), wire::REQUEST_DOMAIN); + self.metrics + .peer_phase(PeerPhase::RequestSign, phase.elapsed()); if encoded.len() > wire::MAX_REQUEST_BYTES { return Err(Error::Capacity("capacity node-log request bytes")); } - let encoded = tokio::time::timeout(Duration::from_secs(10), async { + let phase = std::time::Instant::now(); + let result = tokio::time::timeout(Duration::from_secs(10), async { let response = peer .client .post(format!( @@ -104,8 +115,11 @@ impl Transport { } Ok(bytes) }) - .await - .map_err(transport_error)??; + .await; + self.metrics + .peer_phase(PeerPhase::RoundTrip, phase.elapsed()); + let encoded = result.map_err(transport_error)??; + let phase = std::time::Instant::now(); let body = wire::verify(&encoded, &actual.verifying_key()?, wire::RESPONSE_DOMAIN)?; let reply = wire::Reply::decode(body.as_slice())?; if reply.member != member.as_bytes() || reply.request_digest != digest.as_bytes() { @@ -116,6 +130,8 @@ impl Transport { if reply.status != 0 { return Err(Error::Peer("capacity follower operation failed")); } + self.metrics + .peer_phase(PeerPhase::ReplyVerify, phase.elapsed()); Ok(reply) } } diff --git a/crates/cellule-axum/examples/sql.rs b/crates/cellule-axum/examples/sql.rs index 6ac40198..0420a751 100644 --- a/crates/cellule-axum/examples/sql.rs +++ b/crates/cellule-axum/examples/sql.rs @@ -377,7 +377,9 @@ async fn main() -> ExampleResult<()> { if let Some(config) = &fleet_config && config.index != 0 { - config.serve_follower(layout, code, bind).await?; + config + .serve_follower(layout, code, bind, query_metrics.clone()) + .await?; return Ok(()); } let session = SessionId::from_bytes(*Uuid::now_v7().as_bytes()); @@ -407,7 +409,14 @@ async fn main() -> ExampleResult<()> { if let Some(config) = &fleet_config { fleet_owner = Some( config - .start_owner(layout.clone(), code, session, application_id, &runtime) + .start_owner( + layout.clone(), + code, + session, + application_id, + &runtime, + query_metrics.clone(), + ) .await?, ); owner.endpoint = fleet_owner @@ -478,6 +487,53 @@ async fn main() -> ExampleResult<()> { let typed = ApplicationHandle::::new(client, application, tenant, application_id)?; let router = Router::new() + .route("/debug/metrics", get({ + let metrics = query_metrics.clone(); + let runtime = runtime.clone(); + move || { + let metrics = metrics.clone(); + let runtime = runtime.clone(); + async move { + let mut sample = metrics.window_snapshot(); + let stats = runtime.stats(); + let publication = runtime.publication_progress().await; + sample["publication_progress"] = match publication { + Ok(progress) => serde_json::json!({ + "pending_publications": progress.pending_publications, + "retained_capture_bytes": progress.retained_capture_bytes, + "oldest_unpublished_ms": progress.oldest_unpublished.map(|age| age.as_millis()) + }), + Err(error) => serde_json::json!({"error": error.to_string()}), + }; + sample["node_log_progress"] = match runtime.node_durability() { + Some((_, durability)) => match durability.progress() { + Ok(progress) => serde_json::json!({ + "log_epoch": progress.log_epoch, + "issued_through": progress.issued_through, + "follower_proven_through": progress.follower_proven_through, + "tiered_through": progress.tiered_through, + "pending_object_sequences": progress.pending_object_sequences, + "fleet_active": progress.fleet_active, + "rotating": progress.rotating, + "fenced": progress.fenced + }), + Err(error) => serde_json::json!({"error": error.to_string()}), + }, + None => serde_json::Value::Null, + }; + sample["runtime"] = serde_json::json!({ + "active_cells": stats.active_cells(), + "retained_bytes": stats.retained_bytes(), + "retained_capacity_bytes": stats.retained_capacity_bytes(), + "local_disk_reserved_bytes": stats.local_disk_reserved_bytes(), + "unpublished_node_log_bytes": stats.unpublished_node_log_bytes(), + "worker_jobs": stats.worker_jobs(), "io_slots": stats.io_slots(), + "dirty_jobs": stats.dirty_jobs(), "blocking_jobs": stats.blocking_jobs() + }); + Json(sample) + } + } + })) .route("/orders", post(create_order)) .route("/orders/{id}", get(get_order)) .with_state(ServiceState { @@ -523,6 +579,12 @@ async fn main() -> ExampleResult<()> { Some(owner) => owner.stop().await, None => Ok(()), }; + if let Err(error) = &shutdown { + eprintln!("Runtime drain failed: {error:?}"); + } + if let Err(error) = &fleet_shutdown { + eprintln!("Node enrollment drain failed: {error:?}"); + } result?; shutdown?; fleet_shutdown?; diff --git a/crates/cellule-axum/examples/sql_metrics/capture.rs b/crates/cellule-axum/examples/sql_metrics/capture.rs index 6a4e2a4d..557f02ce 100644 --- a/crates/cellule-axum/examples/sql_metrics/capture.rs +++ b/crates/cellule-axum/examples/sql_metrics/capture.rs @@ -35,6 +35,14 @@ pub(super) struct CaptureMetrics { } impl CaptureMetrics { + pub(super) fn window_snapshot(&self) -> serde_json::Map { + PHASES + .iter() + .zip(&self.phases) + .map(|(name, phase)| (format!("capture_{name}"), phase.raw())) + .collect() + } + pub(super) fn observe(&self, timing: &CaptureTiming, succeeded: bool) { // Preserve CaptureTiming boundaries. A phase not visited by an attempt // contributes zero; this is one bounded histogram set, not Cell labels. diff --git a/crates/cellule-axum/examples/sql_metrics/mod.rs b/crates/cellule-axum/examples/sql_metrics/mod.rs index 85024613..e0dd426e 100644 --- a/crates/cellule-axum/examples/sql_metrics/mod.rs +++ b/crates/cellule-axum/examples/sql_metrics/mod.rs @@ -10,6 +10,7 @@ use std::{ }; mod capture; +mod storage; use capture::CaptureMetrics; #[cfg(test)] @@ -25,13 +26,24 @@ pub(super) struct QueryMetrics { primitive_ns: AtomicU64, writes: WriteMetrics, capture: CaptureMetrics, - storage: [StorageMetrics; StorageOperation::ALL.len()], + storage: storage::Accounting, + peer: [Histogram; PeerPhase::COUNT], host_capacity: serde_json::Value, response_sources: [AtomicU64; 3], + response_elapsed: [Histogram; 3], + response_confirmation: Histogram, + proof_wait: [Histogram; 2], submission_sources: [AtomicU64; 4], log_append_successes: AtomicU64, log_append_failures: AtomicU64, log_append_bytes: AtomicU64, + selected_roots: AtomicU64, + materialized_commits: AtomicU64, + follower_frames: AtomicU64, + follower_sync_calls: AtomicU64, + follower_failures: AtomicU64, + sample_origin: std::sync::OnceLock, + sample_session: std::sync::OnceLock, } // Application-owned instrumentation: fixed histograms, 100-us upper @@ -84,11 +96,71 @@ impl Histogram { }; serde_json::json!({ "count": count, + "total_ns": self.total_ns.load(Ordering::Relaxed), "mean_ms": (count > 0).then(|| self.total_ns.load(Ordering::Relaxed) as f64 / count as f64 / 1_000_000.0), "p50_ms": percentile(50), "p95_ms": percentile(95), "p99_ms": percentile(99), "overflow": buckets[WRITE_BUCKETS - 1], "resolution_us": WRITE_BUCKET_US }) } + + fn raw(&self) -> serde_json::Value { + serde_json::json!({ + "resolution_us": WRITE_BUCKET_US, + "bucket_count": WRITE_BUCKETS, + "total_ns": self.total_ns.load(Ordering::Relaxed), + // Preserve every cumulative count and its original bucket index. + // Serializing hundreds of thousands of empty JSON values on an + // async service thread perturbs the workload being measured. + "nonzero_buckets": self.buckets.iter().enumerate().filter_map(|(index, bucket)| { + let count = bucket.load(Ordering::Relaxed); + (count != 0).then_some((index, count)) + }).collect::>() + }) + } +} + +#[derive(Clone, Copy)] +pub(super) enum PeerPhase { + OwnerEnrollment, + RequestSign, + RoundTrip, + ReplyVerify, + RequestVerify, + ReceiverEnrollment, + DurableAppend, + FollowerWorkerQueue, + FollowerWorker, + FollowerDataSync, +} + +impl PeerPhase { + const COUNT: usize = 10; + const ALL: [Self; Self::COUNT] = [ + Self::OwnerEnrollment, + Self::RequestSign, + Self::RoundTrip, + Self::ReplyVerify, + Self::RequestVerify, + Self::ReceiverEnrollment, + Self::DurableAppend, + Self::FollowerWorkerQueue, + Self::FollowerWorker, + Self::FollowerDataSync, + ]; + fn label(self) -> &'static str { + match self { + Self::OwnerEnrollment => "owner_enrollment", + Self::RequestSign => "request_sign", + Self::RoundTrip => "round_trip", + Self::ReplyVerify => "reply_verify", + Self::RequestVerify => "request_verify", + Self::ReceiverEnrollment => "receiver_enrollment", + Self::DurableAppend => "durable_append", + Self::FollowerWorkerQueue => "follower_worker_queue", + Self::FollowerWorker => "follower_worker", + Self::FollowerDataSync => "follower_data_sync", + } + } } #[derive(Default)] @@ -121,22 +193,18 @@ struct StorageMetrics { } impl StorageObserver for QueryMetrics { - fn started(&self, operation: StorageOperation) { - self.storage[operation.index()] - .started - .fetch_add(1, Ordering::Relaxed); + fn for_object( + &self, + location: Option<&object_store::path::Path>, + ) -> Option> { + Some(self.storage.observer(location)) } + fn started(&self, operation: StorageOperation) { + self.storage.started(operation); + } fn finished(&self, observation: StorageObservation) { - let metrics = &self.storage[observation.operation.index()]; - metrics.duration.observe(observation.duration); - metrics.outcomes[observation.outcome.index()].fetch_add(1, Ordering::Relaxed); - metrics - .bytes_read - .fetch_add(observation.bytes_read, Ordering::Relaxed); - metrics - .bytes_written - .fetch_add(observation.bytes_written, Ordering::Relaxed); + self.storage.finished(observation); } } @@ -145,6 +213,17 @@ fn nanos(elapsed: Duration) -> u64 { } impl CellTelemetry for QueryMetrics { + fn follower_append(&self, timing: cellule_runtime::fleet::telemetry::FollowerAppendTiming) { + self.peer_phase(PeerPhase::FollowerWorkerQueue, timing.worker_queue); + self.peer_phase(PeerPhase::FollowerWorker, timing.worker); + self.peer_phase(PeerPhase::FollowerDataSync, timing.data_sync); + self.follower_frames + .fetch_add(timing.frames, Ordering::Relaxed); + self.follower_sync_calls + .fetch_add(timing.data_sync_calls, Ordering::Relaxed); + self.follower_failures + .fetch_add(u64::from(!timing.succeeded), Ordering::Relaxed); + } fn ltx_capture(&self, timing: &cellule_ltx::CaptureTiming, succeeded: bool) { self.capture.observe(timing, succeeded); } @@ -152,8 +231,8 @@ impl CellTelemetry for QueryMetrics { fn command_response( &self, source: CommandResponseSource, - _elapsed: Duration, - _confirmation: Duration, + elapsed: Duration, + confirmation: Duration, ) { let index = match source { CommandResponseSource::Recorded => 0, @@ -161,6 +240,19 @@ impl CellTelemetry for QueryMetrics { CommandResponseSource::Object => 2, }; self.response_sources[index].fetch_add(1, Ordering::Relaxed); + self.response_elapsed[index].observe(elapsed); + self.response_confirmation.observe(confirmation); + } + fn durability_proof( + &self, + source: cellule_runtime::node::log::DurabilitySource, + waited: Duration, + ) { + let index = match source { + cellule_runtime::node::log::DurabilitySource::Fleet => 0, + cellule_runtime::node::log::DurabilitySource::Object => 1, + }; + self.proof_wait[index].observe(waited); } fn durability_submission(&self, outcome: DurabilitySubmissionOutcome) { let index = match outcome { @@ -184,6 +276,11 @@ impl CellTelemetry for QueryMetrics { self.writes.worker.observe(worker); } fn publication_completed(&self, _cell: cellule_runtime::CellId, timing: PublicationTiming) { + if timing.succeeded { + self.selected_roots.fetch_add(1, Ordering::Relaxed); + self.materialized_commits + .fetch_add(timing.covered_commits, Ordering::Relaxed); + } self.writes.preparation.observe(timing.preparation); self.writes.authority.observe(timing.authority); self.writes.publication.observe(timing.total); @@ -232,6 +329,57 @@ impl CellTelemetry for QueryMetrics { } impl QueryMetrics { + pub(super) fn peer_phase(&self, phase: PeerPhase, elapsed: Duration) { + self.peer[phase as usize].observe(elapsed); + } + + pub(super) async fn enrollment( + &self, + receiver: bool, + future: impl std::future::Future, + ) -> T { + let started = std::time::Instant::now(); + let result = storage::enrollment(receiver, future).await; + self.peer_phase( + if receiver { + PeerPhase::ReceiverEnrollment + } else { + PeerPhase::OwnerEnrollment + }, + started.elapsed(), + ); + result + } + + pub(super) fn window_snapshot(&self) -> serde_json::Value { + let mut value = self.snapshot(); + let mut histograms = serde_json::Map::new(); + for (label, histogram) in [ + ("actor_queue", &self.writes.queue), + ("worker", &self.writes.worker), + ("publication", &self.writes.publication), + ("compaction", &self.writes.compaction), + ("dirty_admission", &self.writes.dirty_admission), + ] { + histograms.insert(label.into(), histogram.raw()); + } + for phase in PeerPhase::ALL { + histograms.insert(phase.label().into(), self.peer[phase as usize].raw()); + } + for (name, histogram) in [ + ("response_recorded", &self.response_elapsed[0]), + ("response_fleet", &self.response_elapsed[1]), + ("response_object", &self.response_elapsed[2]), + ("response_confirmation", &self.response_confirmation), + ("proof_fleet", &self.proof_wait[0]), + ("proof_object", &self.proof_wait[1]), + ] { + histograms.insert(name.into(), histogram.raw()); + } + histograms.extend(self.capture.window_snapshot()); + value["histograms"] = histograms.into(); + value + } pub(super) fn new(host: &cellule_ltx::Host) -> Self { Self { host_capacity: serde_json::json!({ @@ -256,33 +404,52 @@ impl QueryMetrics { Some(total.load(Ordering::Relaxed) as f64 / count as f64 / 1000.0) } }; + // Derive aggregate counters from this one family snapshot. Reading + // separate live totals before families introduces false reconciliation + // gaps while callbacks complete. Duration histograms remain independent. + let storage_families = self.storage.snapshot(); + let sum = |operation: &str, field: &str, outcome: Option<&str>| -> u64 { + storage_families + .as_object() + .into_iter() + .flat_map(|families| families.values()) + .map(|family| { + let value = &family[operation][field]; + outcome.map_or_else( + || value.as_u64().unwrap_or(0), + |outcome| value[outcome].as_u64().unwrap_or(0), + ) + }) + .sum() + }; let storage: serde_json::Map = StorageOperation::ALL .iter() .map(|operation| { - let metrics = &self.storage[operation.index()]; + let metrics = &self.storage.total[operation.index()]; let outcomes: serde_json::Map = StorageOutcome::ALL .iter() .map(|outcome| { ( outcome.label().into(), - metrics.outcomes[outcome.index()] - .load(Ordering::Relaxed) - .into(), + sum(operation.label(), "outcomes", Some(outcome.label())).into(), ) }) .collect(); ( operation.label().into(), serde_json::json!({ - "started": metrics.started.load(Ordering::Relaxed), + "started": sum(operation.label(), "started", None), "duration": metrics.duration.snapshot(), "outcomes": outcomes, - "bytes_read": metrics.bytes_read.load(Ordering::Relaxed), - "bytes_written": metrics.bytes_written.load(Ordering::Relaxed) + "bytes_read": sum(operation.label(), "bytes_read", None), + "bytes_written": sum(operation.label(), "bytes_written", None) }), ) }) .collect(); serde_json::json!({ + "schema_version": 3, + "sample_session": self.sample_session.get_or_init(uuid::Uuid::now_v7).to_string(), + "sample_elapsed_ns": nanos(self.sample_origin.get_or_init(std::time::Instant::now).elapsed()), "host_capacity": self.host_capacity, "response_sources": { "recorded": self.response_sources[0].load(Ordering::Relaxed), @@ -300,7 +467,14 @@ impl QueryMetrics { "failures": self.log_append_failures.load(Ordering::Relaxed), "bytes": self.log_append_bytes.load(Ordering::Relaxed) }, + "follower_append": { + "input_frames": self.follower_frames.load(Ordering::Relaxed), + "data_sync_calls": self.follower_sync_calls.load(Ordering::Relaxed), + "failures": self.follower_failures.load(Ordering::Relaxed) + }, "storage_operations": storage, + "storage_families": storage_families, + "peer_phases": PeerPhase::ALL.iter().map(|phase| (phase.label().to_owned(), self.peer[*phase as usize].snapshot())).collect::>(), "queries": queries, "failures": self.failures.load(Ordering::Relaxed), "mean_actor_queue_us": mean_us(&self.queue_ns, queries), @@ -323,6 +497,8 @@ impl QueryMetrics { "dirty_admission": self.writes.dirty_admission.snapshot(), "recovery_admission": self.writes.recovery_admission.snapshot(), "publication_failures": self.writes.publication_failures.load(Ordering::Relaxed), + "selected_roots": self.selected_roots.load(Ordering::Relaxed), + "materialized_commits": self.materialized_commits.load(Ordering::Relaxed), "uploaded_objects": self.writes.objects.load(Ordering::Relaxed), "uploaded_bytes": self.writes.bytes.load(Ordering::Relaxed) } diff --git a/crates/cellule-axum/examples/sql_metrics/storage.rs b/crates/cellule-axum/examples/sql_metrics/storage.rs new file mode 100644 index 00000000..fb30336f --- /dev/null +++ b/crates/cellule-axum/examples/sql_metrics/storage.rs @@ -0,0 +1,166 @@ +//! Fixed object families chosen by the application, without identity labels. +use super::*; +use std::sync::Arc; + +const FAMILIES: [&str; 6] = [ + "immutable", + "cell_authority", + "node_authority", + "other", + "owner_enrollment", + "receiver_enrollment", +]; + +tokio::task_local! { + static ENROLLMENT_FAMILY: usize; +} + +pub(super) async fn enrollment( + receiver: bool, + future: impl std::future::Future, +) -> T { + ENROLLMENT_FAMILY + .scope(if receiver { 5 } else { 4 }, future) + .await +} + +#[derive(Default)] +struct Counts { + started: AtomicU64, + outcomes: [AtomicU64; StorageOutcome::ALL.len()], + bytes_read: AtomicU64, + bytes_written: AtomicU64, +} + +struct Family { + total: Arc<[StorageMetrics; StorageOperation::ALL.len()]>, + counts: [Counts; StorageOperation::ALL.len()], +} + +impl StorageObserver for Family { + fn started(&self, operation: StorageOperation) { + self.total[operation.index()] + .started + .fetch_add(1, Ordering::Relaxed); + self.counts[operation.index()] + .started + .fetch_add(1, Ordering::Relaxed); + } + + fn finished(&self, observation: StorageObservation) { + record(&self.total[observation.operation.index()], observation); + let counts = &self.counts[observation.operation.index()]; + counts.outcomes[observation.outcome.index()].fetch_add(1, Ordering::Relaxed); + counts + .bytes_read + .fetch_add(observation.bytes_read, Ordering::Relaxed); + counts + .bytes_written + .fetch_add(observation.bytes_written, Ordering::Relaxed); + } +} + +pub(super) struct Accounting { + pub total: Arc<[StorageMetrics; StorageOperation::ALL.len()]>, + families: [Arc; FAMILIES.len()], +} + +impl Default for Accounting { + fn default() -> Self { + let total = Arc::new(std::array::from_fn(|_| StorageMetrics::default())); + Self { + families: std::array::from_fn(|_| { + Arc::new(Family { + total: total.clone(), + counts: std::array::from_fn(|_| Counts::default()), + }) + }), + total, + } + } +} + +impl Accounting { + pub fn started(&self, operation: StorageOperation) { + self.families[3].started(operation); + } + pub fn finished(&self, observation: StorageObservation) { + self.families[3].finished(observation); + } + + pub fn observer( + &self, + location: Option<&object_store::path::Path>, + ) -> Arc { + let index = location.map_or(3, |path| { + let name = path.as_ref(); + if [".ltx", ".index", ".dir", ".root", ".bundle", ".pack"] + .iter() + .any(|suffix| name.ends_with(suffix)) + { + 0 + } else if name + .split('/') + .any(|part| matches!(part, "nodes" | "node-logs")) + { + // Select once before dispatch. The returned observer owns the + // role through streamed completion, beyond this task scope. + ENROLLMENT_FAMILY.try_with(|index| *index).unwrap_or(2) + } else if name.split('/').any(|part| part == "cells") { + 1 + } else { + 3 + } + }); + self.families[index].clone() + } + + pub fn snapshot(&self) -> serde_json::Value { + let families: serde_json::Map = FAMILIES + .iter() + .zip(&self.families) + .map(|(label, family)| { + let operations: serde_json::Map = StorageOperation::ALL + .iter() + .map(|operation| { + let counts = &family.counts[operation.index()]; + let outcomes: serde_json::Map = + StorageOutcome::ALL + .iter() + .map(|outcome| { + ( + outcome.label().into(), + counts.outcomes[outcome.index()] + .load(Ordering::Relaxed) + .into(), + ) + }) + .collect(); + ( + operation.label().into(), + serde_json::json!({ + "started": counts.started.load(Ordering::Relaxed), + "outcomes": outcomes, + "bytes_read": counts.bytes_read.load(Ordering::Relaxed), + "bytes_written": counts.bytes_written.load(Ordering::Relaxed), + }), + ) + }) + .collect(); + ((*label).into(), operations.into()) + }) + .collect(); + families.into() + } +} + +pub(super) fn record(metrics: &StorageMetrics, observation: StorageObservation) { + metrics.duration.observe(observation.duration); + metrics.outcomes[observation.outcome.index()].fetch_add(1, Ordering::Relaxed); + metrics + .bytes_read + .fetch_add(observation.bytes_read, Ordering::Relaxed); + metrics + .bytes_written + .fetch_add(observation.bytes_written, Ordering::Relaxed); +} diff --git a/crates/cellule-axum/examples/sql_metrics/tests.rs b/crates/cellule-axum/examples/sql_metrics/tests.rs index b51bd204..fa673096 100644 --- a/crates/cellule-axum/examples/sql_metrics/tests.rs +++ b/crates/cellule-axum/examples/sql_metrics/tests.rs @@ -1,5 +1,179 @@ use super::*; +#[test] +fn sparse_export_preserves_cumulative_histogram_indices_and_overflow() { + let histogram = Histogram::default(); + histogram.observe(Duration::from_micros(201)); + histogram.observe(Duration::from_micros(201)); + histogram.observe(Duration::from_secs(3)); + let raw = histogram.raw(); + assert_eq!(raw["bucket_count"], WRITE_BUCKETS); + assert_eq!(raw["resolution_us"], 100); + assert_eq!( + raw["nonzero_buckets"], + serde_json::json!([[3, 2], [WRITE_BUCKETS - 1, 1]]) + ); + assert_eq!(raw["total_ns"], 3_000_402_000_u64); + assert!(raw.get("buckets").is_none()); +} + +#[test] +fn response_and_proof_timings_keep_ack_latency_separate_from_materialization() { + let metrics = QueryMetrics::default(); + metrics.command_response( + CommandResponseSource::Fleet, + Duration::from_millis(2), + Duration::from_millis(1), + ); + metrics.durability_proof( + cellule_runtime::node::log::DurabilitySource::Fleet, + Duration::from_millis(1), + ); + metrics.durability_proof( + cellule_runtime::node::log::DurabilitySource::Object, + Duration::from_millis(50), + ); + let sample = metrics.window_snapshot(); + assert_eq!( + sample["histograms"]["response_fleet"]["nonzero_buckets"], + serde_json::json!([[20, 1]]) + ); + assert_eq!( + sample["histograms"]["proof_fleet"]["nonzero_buckets"], + serde_json::json!([[10, 1]]) + ); + assert_eq!( + sample["histograms"]["proof_object"]["nonzero_buckets"], + serde_json::json!([[500, 1]]) + ); + assert_eq!(sample["response_sources"]["fleet"], 1); + assert_eq!(sample["response_sources"]["object"], 0); +} + +#[tokio::test] +async fn enrollment_roles_survive_stream_completion_without_leaking_into_other_tasks() { + use object_store::{ObjectStoreExt as _, memory::InMemory, path::Path}; + let metrics = std::sync::Arc::new(QueryMetrics::default()); + let store = cellule_store::Store::new(std::sync::Arc::new(InMemory::new())) + .with_storage_observer(metrics.clone()); + let path = Path::from("fixture/nodes/session.json"); + store + .inner() + .put(&path, bytes::Bytes::from_static(b"enrollment").into()) + .await + .unwrap(); + let (owner, receiver) = tokio::join!( + metrics.enrollment(false, store.inner().get(&path)), + metrics.enrollment(true, store.inner().get(&path)), + ); + // Both task scopes have ended, but streamed accounting retains its role. + owner.unwrap().bytes().await.unwrap(); + receiver.unwrap().bytes().await.unwrap(); + store + .inner() + .get(&path) + .await + .unwrap() + .bytes() + .await + .unwrap(); + let snapshot = metrics.snapshot(); + for role in ["owner_enrollment", "receiver_enrollment", "node_authority"] { + assert_eq!( + snapshot["storage_families"][role]["get"]["outcomes"]["success"], + 1 + ); + } + assert_eq!( + snapshot["storage_operations"]["get"]["outcomes"]["success"], + 3 + ); +} + +#[test] +fn commands_per_root_counts_only_confirmed_publications() { + let metrics = QueryMetrics::default(); + let timing = PublicationTiming { + queue_wait: Duration::ZERO, + preparation: Duration::ZERO, + authority: Duration::ZERO, + total: Duration::ZERO, + succeeded: true, + commit_sequence: 12, + covered_commits: 12, + }; + metrics.publication_completed(cellule_runtime::CellId::from_bytes([1; 32]), timing); + metrics.publication_completed( + cellule_runtime::CellId::from_bytes([1; 32]), + PublicationTiming { + succeeded: false, + covered_commits: 100, + ..timing + }, + ); + let snapshot = metrics.snapshot(); + assert_eq!(snapshot["writes"]["selected_roots"], 1); + assert_eq!(snapshot["writes"]["materialized_commits"], 12); + assert_eq!(snapshot["writes"]["publication_failures"], 1); +} + +#[tokio::test] +async fn fixed_storage_families_reconcile_after_stream_completion() { + use object_store::{ObjectStoreExt as _, memory::InMemory, path::Path}; + let metrics = std::sync::Arc::new(QueryMetrics::default()); + let store = cellule_store::Store::new(std::sync::Arc::new(InMemory::new())) + .with_storage_observer(metrics.clone()); + for path in [ + "fixture/cells/v1/app/cells/x/inc/y/object.root", + "fixture/cells/v1/app/cells/x/control.json", + "fixture/cells/v1/app/nodes/session.json", + "fixture/unclassified", + ] { + let path = Path::from(path); + store + .inner() + .put(&path, bytes::Bytes::from_static(b"verified").into()) + .await + .unwrap(); + assert_eq!( + store + .inner() + .get(&path) + .await + .unwrap() + .bytes() + .await + .unwrap(), + "verified" + ); + } + let snapshot = metrics.snapshot(); + for operation in ["put", "get"] { + let sum: u64 = snapshot["storage_families"] + .as_object() + .unwrap() + .values() + .map(|family| family[operation]["outcomes"]["success"].as_u64().unwrap()) + .sum(); + assert_eq!( + sum, + snapshot["storage_operations"][operation]["outcomes"]["success"] + ); + assert_eq!(sum, 4); + } + metrics.peer_phase(PeerPhase::RoundTrip, Duration::from_micros(350)); + let raw = metrics.window_snapshot(); + assert_eq!( + raw["histograms"]["round_trip"]["nonzero_buckets"], + serde_json::json!([[4, 1]]) + ); + assert_eq!( + raw["histograms"]["round_trip"]["bucket_count"], + WRITE_BUCKETS + ); + assert_eq!(raw["histograms"]["round_trip"]["total_ns"], 350_000); +} + #[test] fn real_capture_and_capacity_failure_reach_the_application_snapshot() { let directory = tempfile::TempDir::new().unwrap(); diff --git a/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/writer_tests/successors/mod.rs b/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/writer_tests/successors/mod.rs index f7800d8a..250b939f 100644 --- a/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/writer_tests/successors/mod.rs +++ b/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/writer_tests/successors/mod.rs @@ -401,7 +401,7 @@ async fn complete_original_successors_read_actual_origin_despite_live_native_sql let objects = inputs.replica.reachable_objects(&root).await.unwrap(); let missing = objects .iter() - .find(|row| row.kind == CellObjectKind::Ltx) + .find(|row| matches!(row.kind, CellObjectKind::Ltx | CellObjectKind::Packed)) .unwrap(); let captured = fixture.original.capture().await.unwrap(); let target = captured diff --git a/crates/cellule-ltx/docs/README.md b/crates/cellule-ltx/docs/README.md index 03dbd511..9cf7c488 100644 --- a/crates/cellule-ltx/docs/README.md +++ b/crates/cellule-ltx/docs/README.md @@ -584,10 +584,11 @@ generation reset; it never truncates uncaptured pages. Standalone synchronous capture retains its ordinary checkpoint threshold. - **Overlap.** Immutable preparation overlaps independent uploads without - weakening the root gate: each LTX body uploads alongside its index, changed - directory nodes upload concurrently, initial directory construction streams - nodes in eight-object waves, and the root document uploads alongside its - segment pages. + weakening the root gate: small segment/index pairs share one authenticated + packed object and a bounded directory leaf lives in its root. Larger bodies, + indexes and directories retain parallel streaming uploads. Root documents + upload alongside their segment pages. The [current root format](packed-root-format.md) + defines the exact bounds and development cutover. - **Concurrency ceiling.** Up to four captured segments and eight small metadata objects progress concurrently; the shared host I/O permits remain the process-wide request ceiling. @@ -838,9 +839,10 @@ These are per-operation correctness bounds, not an RSS quota. `take_publication_cost` report the exact object count and bytes per root, so a host can budget object-store cost per command instead of inferring it from the database size. -- **Measured cost.** The local measurement frozen in `tests/cell/roots/lifecycle.rs` - is five objects per small append (about 7 KiB for a 4 KiB payload); - provider-scale cost distributions remain outstanding. +- **Measured cost.** `tests/cell/roots/prepare_cost.rs` verifies two immutable + PUTs per small selected root through 32 successive cuts, with a predecessor + origin check. Runtime lineage and fenced control selection add two PUTs. + Larger roots and maintenance retain their own measured operation counts. - **Plan memory.** Each live `VerifiedPlan` retains one reconstructed database image, bounded by `max_database_bytes`, plus its checksum state and segment metadata. diff --git a/crates/cellule-ltx/docs/packed-root-format.md b/crates/cellule-ltx/docs/packed-root-format.md new file mode 100644 index 00000000..bfbd85e5 --- /dev/null +++ b/crates/cellule-ltx/docs/packed-root-format.md @@ -0,0 +1,73 @@ +# Bounded packed dependencies + +Cell root JSON version 2 is the current development format. It replaces version +1 atomically; there is no legacy decoder or dual-write path. Native LTX bytes, +node frame formats, Cell fencing, accumulating lineage and authority selection +remain their existing contracts. + +## Small segment object + +An object ending in `.pack` contains exactly one original LTX segment and its +fixed-width authenticated page index. Its total length is at most 256 KiB. +Larger segments use separate `.ltx` and `.index` objects. The complete packed +object is named by its BLAKE3 digest in the producing Cell incarnation's ordinary +immutable object directory. + +| Bytes | Encoding | +| --- | --- | +| 0–7 | ASCII `CRBPACK1` | +| 8–15 | Original LTX length, unsigned big-endian u64 | +| 16–23 | Fixed-width index length, unsigned big-endian u64 | +| 24–31 | Eight zero reserved bytes | +| 32–63 | Original LTX BLAKE3 digest | +| 64 onward | Exact original LTX bytes, then exact fixed-width index bytes | + +The segment descriptor has a required `packed` boolean. A packed descriptor +pins the complete object digest, a body offset of exactly 64, the original LTX +length and digest, and the index length and digest. The index starts immediately +after the body. Checked arithmetic rejects overflow, overlap, foreign offsets, +and any total exceeding the bound. Separate and bundled descriptors have +`packed: false` and preserve their ordinary extents. + +Preparation verifies the source size and hash and retains its pinned handle and +shared index. Upload freezes and rechecks the complete packed bytes while holding +host I/O admission. Waiting proposals retain no additional packed body buffer. +Small compaction outputs use the same representation. Sparse reads continue +verifying individual frame hashes from the authenticated directory. Compaction +authenticates the complete packed source, including bytes outside page frames. + +## Inline directory leaf + +The root has a canonical `directory_inline` field containing lowercase hex or +`null`. It may contain one directory leaf of at most 2 KiB, only at height zero, +with BLAKE3 matching `directory_digest`. All leaf checks remain mandatory: +ordered coverage, checksum, page size, frame hashes and body extent bounds. +The entire root remains bounded by 32 KiB. If inline data would exceed that +bound, the leaf uses the existing `.dir` object path. + +Cached predecessor root bytes still require origin presence. Uncached inventory +authenticates the root and every packed object; an inline leaf requires no +separate directory object. Backup pinning, recovery-prefix verification and +collection consume this exact inventory. Collection recognizes `.pack` objects +under the same quiescence and grace rules as other Cell objects. + +Compaction fetches each small packed input once, verifies its complete header, +body and index, then writes both admitted scratch extents from those bytes. +The transfer window and host I/O slot bound retained memory through dispatched +writes and cancellation. Large objects keep separate bounded streams. + +## Evidence and cutover + +`tests/cell/roots/packed.rs` validates the header and original bytes, corrupts +header/body/index bytes, truncates and deletes the object, and rejects cross-Cell +scope and invalid extents. Coalescing, compaction, sparse activation, restore, +response-loss reconciliation and missing-origin tests cover both representations. +`tests/cell/roots/prepare_cost.rs` requires two immutable PUTs for a small root. +Runtime lineage and fenced control selection add two successful PUTs. + +Follow the workspace [development format policy](../../cellule-runtime/docs/storage.md#format-policy). +All producers and consumers must deploy the current format together, including +backup, recovery and collection workers. Qualification uses fresh isolated +prefixes. An older binary cannot read version 2 roots; rollback requires a +verified logical export/rebuild or recreating disposable development fixtures. +Do not run mixed root-format binaries against one prefix. diff --git a/crates/cellule-ltx/src/cell_layout.rs b/crates/cellule-ltx/src/cell_layout.rs index 0321e1c9..bf50feb5 100644 --- a/crates/cellule-ltx/src/cell_layout.rs +++ b/crates/cellule-ltx/src/cell_layout.rs @@ -25,6 +25,8 @@ pub enum CellObjectKind { Root, /// Recovery bundle. Bundle, + /// Bounded segment and authenticated fixed-width index in one object. + Packed, } impl CellObjectKind { @@ -35,6 +37,7 @@ impl CellObjectKind { Self::Directory => "dir", Self::Root => "root", Self::Bundle => "bundle", + Self::Packed => "pack", } } } diff --git a/crates/cellule-ltx/src/replica/compaction/mod.rs b/crates/cellule-ltx/src/replica/compaction/mod.rs index aebfebf9..60d81d55 100644 --- a/crates/cellule-ltx/src/replica/compaction/mod.rs +++ b/crates/cellule-ltx/src/replica/compaction/mod.rs @@ -96,15 +96,14 @@ async fn prepare_root( append: Option, ) -> Result { let selected = &graph.descriptors[range.clone()]; - // The authenticated streams have separate scratch files. Both must finish - // before the merge, but neither depends on the other's transfer. - let (spooled, body_inputs) = futures_util::future::join( - spool_indexes(replica, selected, &files.scratch, &files.original_indexes), - spool_selected_bodies(replica, selected, &files.scratch, &files.original_bodies), + let (spooled, body_inputs) = spool_selected( + replica, + selected, + &files.scratch, + &files.original_bodies, + &files.original_indexes, ) - .await; - let spooled = spooled?; - let body_inputs = body_inputs?; + .await?; let artifacts = write_compacted(replica, spooled, &body_inputs, &files).await?; let first = selected.first().ok_or(LtxError::TxNotAvailable)?; @@ -122,12 +121,55 @@ async fn prepare_root( let descriptor = SegmentDescriptor::native(info, artifacts.index.digest, artifacts.index.length) .with_level(level); + let packed_segment = if super::packed::HEADER_BYTES + .checked_add(artifacts.ltx.length) + .and_then(|n| n.checked_add(artifacts.index.length)) + .is_some_and(|n| n <= super::upload::SINGLE_PUT_BYTES) + { + let scratch = Arc::clone(&files.scratch); + let path = files.compacted_index.clone(); + let length = artifacts.index.length as usize; + let index = replica + .host + .run(move || { + // A dispatched read owns scratch through cancellation. Opening + // read-only also preserves the source-read error boundary. + let mut file = scratch.host.filesystem.open(&path)?; + file.read_exact_at(0, length) + }) + .await??; + let source = super::upload::PinnedCapture::open( + &replica.host, + files.compacted_ltx.clone(), + artifacts.ltx.length, + ) + .await?; + Some( + super::packed::freeze( + replica, + super::PreparedSegment { + descriptor: descriptor.clone(), + index: bytes::Bytes::from(index), + body: super::AppendBody::Native(source), + }, + ) + .await?, + ) + } else { + None + }; + let descriptor = packed_segment + .as_ref() + .map_or(descriptor, |segment| segment.descriptor.clone()); descriptor.validate_published(replica.limits)?; let mut descriptors = graph.descriptors.clone(); descriptors.splice(range.clone(), [descriptor.clone()]); replica.validate_chain(&descriptors, base.position)?; let dependency_uploads = async { + if let Some(segment) = packed_segment { + return replica.upload_prepared_segment(segment).await; + } let (body, index) = futures_util::future::join( upload( replica, @@ -163,13 +205,14 @@ async fn prepare_root( let endpoint = descriptors.last().ok_or(LtxError::LTXCorrupted)?; let page_size = endpoint.info.page_size; let database_pages = endpoint.info.database_pages; - let retain_leaf = append.as_ref().is_some_and(|append| { - graph.document.directory_height == 0 - && append - .inputs - .last() - .is_some_and(|input| directory::fits_leaf(input.info.database_pages)) - }); + let retain_leaf = graph.document.directory_height == 0 + && append.as_ref().is_none_or(|append| { + graph.document.directory_height == 0 + && append + .inputs + .last() + .is_some_and(|input| directory::fits_leaf(input.info.database_pages)) + }); let directory = directory::relocate_and_upload( replica, &graph, @@ -187,6 +230,7 @@ async fn prepare_root( aggregate: graph.aggregate, directory_digest: directory.root_digest(), directory_height: directory.height(), + directory_inline: None, page_size, database_pages, inherited_segment_pages: graph.document.segment_pages.clone(), diff --git a/crates/cellule-ltx/src/replica/compaction/source.rs b/crates/cellule-ltx/src/replica/compaction/source.rs index 6b65b430..a9928f6e 100644 --- a/crates/cellule-ltx/src/replica/compaction/source.rs +++ b/crates/cellule-ltx/src/replica/compaction/source.rs @@ -3,72 +3,58 @@ use super::*; use futures_util::FutureExt as _; -pub(super) async fn spool_selected_bodies( +pub(super) async fn spool_selected( replica: &CellReplica, descriptors: &[SegmentDescriptor], scratch: &Arc, - destination: &Path, -) -> Result> { + body_destination: &Path, + index_destination: &Path, +) -> Result<(Vec, Vec)> { let mut planned = Vec::with_capacity(descriptors.len()); - let mut total_bytes = 0_u64; + let (mut body_bytes, mut index_bytes) = (0_u64, 0_u64); for descriptor in descriptors { - let start = total_bytes; - total_bytes = total_bytes + if descriptor.index_length == 0 + || descriptor.index_length % crate::paged::ENTRY_BYTES as u64 != 0 + { + return Err(LtxError::LTXCorrupted); + } + planned.push((descriptor.clone(), body_bytes, index_bytes)); + body_bytes = body_bytes .checked_add(descriptor.info.size_bytes) .ok_or(LtxError::Limit(crate::LimitKind::CompactionBodySpool))?; - planned.push((descriptor.clone(), start)); + index_bytes = index_bytes + .checked_add(descriptor.index_length) + .ok_or(LtxError::Limit(crate::LimitKind::CompactionIndexSpool))?; } - let results = stream::iter(planned.into_iter().enumerate().map( - |(order, (descriptor, output_start))| { + |(order, (descriptor, body_start, index_start))| { async move { - let mut file = scratch.open(destination).await?; - let source_start = descriptor.offset(); - let source_end = source_start - .checked_add(descriptor.info.size_bytes) - .ok_or(LtxError::LTXCorrupted)?; - let path = replica.layout.incarnation_object_path( - &replica.cell, - &replica.incarnation, - &descriptor.object_digest(), - descriptor.object_kind(), - ); - let mut source_offset = source_start; - let mut hasher = blake3::Hasher::new(); - while source_offset < source_end { - let next = source_offset - .checked_add((source_end - source_offset).min(FRAME_READ_BYTES)) - .ok_or(LtxError::LTXCorrupted)?; - let _permit = replica.host.io_permit().await?; - let bytes = replica - .layout - .store() - .range_get(&path, source_offset..next) - .await?; - drop(_permit); - if bytes.len() as u64 != next - source_offset { - return Err(LtxError::ChecksumMismatch); - } - hasher.update(&bytes); - let output_offset = output_start - .checked_add(source_offset - source_start) - .ok_or(LtxError::Limit(crate::LimitKind::CompactionBodySpool))?; - file = replica - .host - .run(move || { - file.write_all_at(output_offset, &bytes)?; - Ok::<_, LtxError>(file) - }) - .await??; - source_offset = next; + if descriptor.object_kind() == CellObjectKind::Packed { + return spool_packed( + replica, + descriptor, + scratch, + body_destination, + index_destination, + body_start, + index_start, + ) + .await; } - if *hasher.finalize().as_bytes() != descriptor.info.blake3 { - return Err(LtxError::ChecksumMismatch); - } - Ok(BodySpoolInput { - descriptor, - start: output_start, - }) + // Separate large objects retain bounded streaming. Join both + // transfers even on error so dispatched file jobs keep ownership. + let (index, body) = futures_util::future::join( + spool_index( + replica, + descriptor.clone(), + scratch, + index_destination, + index_start, + ), + spool_body(replica, descriptor, scratch, body_destination, body_start), + ) + .await; + Ok((index?, body?)) } .map(move |result| (order, result)) }, @@ -76,94 +62,188 @@ pub(super) async fn spool_selected_bodies( .buffer_unordered(SEGMENT_TRANSFER_CONCURRENCY) .collect::>() .await; - let spooled = ordered_results(results)?; - sync_spool(scratch, destination, total_bytes).await?; - Ok(spooled) + let (indexes, bodies) = ordered_results(results)?.into_iter().unzip(); + let (body_sync, index_sync) = futures_util::future::join( + sync_spool(scratch, body_destination, body_bytes), + sync_spool(scratch, index_destination, index_bytes), + ) + .await; + body_sync?; + index_sync?; + Ok((indexes, bodies)) } -pub(super) async fn spool_indexes( +#[allow( + clippy::too_many_arguments, + reason = "one verified object supplies two independently bounded spool extents" +)] +async fn spool_packed( replica: &CellReplica, - descriptors: &[SegmentDescriptor], + descriptor: SegmentDescriptor, + scratch: &Arc, + body_destination: &Path, + index_destination: &Path, + body_start: u64, + index_start: u64, +) -> Result<(SpoolInput, BodySpoolInput)> { + let path = replica.layout.incarnation_object_path( + &replica.cell, + &replica.incarnation, + &descriptor.object_digest(), + CellObjectKind::Packed, + ); + let permit = replica.host.io_permit().await?; + let (bytes, _) = replica + .layout + .store() + .get_with_etag_bounded(&path, super::super::upload::SINGLE_PUT_BYTES) + .await?; + super::super::packed::verify(&bytes, &descriptor)?; + let body_offset = descriptor.offset() as usize; + let index_offset = (descriptor.offset() + descriptor.info.size_bytes) as usize; + let body = bytes.slice(body_offset..index_offset); + let index = bytes.slice(index_offset..); + let mut body_file = scratch.open(body_destination).await?; + let mut index_file = scratch.open(index_destination).await?; + replica + .host + .run(move || { + // One I/O slot bounds the pack's memory until both dispatched writes + // finish, including cancellation. ScratchFile retains its file owner. + let _permit = permit; + body_file.write_all_at(body_start, &body)?; + index_file.write_all_at(index_start, &index)?; + Ok::<_, LtxError>(()) + }) + .await??; + let length = descriptor.index_length; + Ok(( + SpoolInput { + descriptor: descriptor.clone(), + start: index_start, + length, + }, + BodySpoolInput { + descriptor, + start: body_start, + }, + )) +} + +async fn spool_body( + replica: &CellReplica, + descriptor: SegmentDescriptor, scratch: &Arc, destination: &Path, -) -> Result> { - let mut planned = Vec::with_capacity(descriptors.len()); - let mut total_bytes = 0_u64; - for descriptor in descriptors { - if descriptor.index_length == 0 - || descriptor.index_length % crate::paged::ENTRY_BYTES as u64 != 0 - { - return Err(LtxError::LTXCorrupted); + output_start: u64, +) -> Result { + let mut file = scratch.open(destination).await?; + let source_start = descriptor.offset(); + let source_end = source_start + .checked_add(descriptor.info.size_bytes) + .ok_or(LtxError::LTXCorrupted)?; + let path = replica.layout.incarnation_object_path( + &replica.cell, + &replica.incarnation, + &descriptor.object_digest(), + descriptor.object_kind(), + ); + let mut source_offset = source_start; + let mut hasher = blake3::Hasher::new(); + while source_offset < source_end { + let next = source_offset + .checked_add((source_end - source_offset).min(FRAME_READ_BYTES)) + .ok_or(LtxError::LTXCorrupted)?; + let permit = replica.host.io_permit().await?; + let bytes = replica + .layout + .store() + .range_get(&path, source_offset..next) + .await?; + drop(permit); + if bytes.len() as u64 != next - source_offset { + return Err(LtxError::ChecksumMismatch); } - let start = total_bytes; - total_bytes = total_bytes - .checked_add(descriptor.index_length) - .ok_or(LtxError::Limit(crate::LimitKind::CompactionIndexSpool))?; - planned.push((descriptor.clone(), start)); + hasher.update(&bytes); + let output_offset = output_start + .checked_add(source_offset - source_start) + .ok_or(LtxError::Limit(crate::LimitKind::CompactionBodySpool))?; + file = replica + .host + .run(move || { + file.write_all_at(output_offset, &bytes)?; + Ok::<_, LtxError>(file) + }) + .await??; + source_offset = next; } + if *hasher.finalize().as_bytes() != descriptor.info.blake3 { + return Err(LtxError::ChecksumMismatch); + } + Ok(BodySpoolInput { + descriptor, + start: output_start, + }) +} - let results = stream::iter(planned.into_iter().enumerate().map( - |(order, (descriptor, output_start))| { - async move { - let mut file = scratch.open(destination).await?; - let path = replica.layout.incarnation_object_path( - &replica.cell, - &replica.incarnation, - &descriptor.index_digest, - CellObjectKind::Index, - ); - let mut source_offset = 0_u64; - let mut hasher = blake3::Hasher::new(); - let mut validator = crate::paged::IndexValidator::new(&descriptor.info); - while source_offset < descriptor.index_length { - let length = (descriptor.index_length - source_offset).min(INDEX_READ_BYTES); - let _permit = replica.host.io_permit().await?; - let bytes = replica - .layout - .store() - .range_get(&path, source_offset..source_offset + length) - .await?; - drop(_permit); - if bytes.len() as u64 != length { - return Err(LtxError::ChecksumMismatch); - } - hasher.update(&bytes); - let output_offset = output_start - .checked_add(source_offset) - .ok_or(LtxError::Limit(crate::LimitKind::CompactionIndexSpool))?; - let returned = replica - .host - .run(move || { - for entry in bytes.as_chunks::<{ crate::paged::ENTRY_BYTES }>().0 { - validator.validate(crate::paged::decode_index_entry(entry)?)?; - } - file.write_all_at(output_offset, &bytes)?; - Ok::<_, LtxError>((file, validator)) - }) - .await??; - file = returned.0; - validator = returned.1; - source_offset += length; - } - if *hasher.finalize().as_bytes() != descriptor.index_digest { - return Err(LtxError::ChecksumMismatch); +async fn spool_index( + replica: &CellReplica, + descriptor: SegmentDescriptor, + scratch: &Arc, + destination: &Path, + output_start: u64, +) -> Result { + let mut file = scratch.open(destination).await?; + let (object, kind, index_start) = descriptor.index_extent(); + let path = + replica + .layout + .incarnation_object_path(&replica.cell, &replica.incarnation, &object, kind); + let mut source_offset = 0_u64; + let mut hasher = blake3::Hasher::new(); + let mut validator = crate::paged::IndexValidator::new(&descriptor.info); + while source_offset < descriptor.index_length { + let length = (descriptor.index_length - source_offset).min(INDEX_READ_BYTES); + let permit = replica.host.io_permit().await?; + let bytes = replica + .layout + .store() + .range_get( + &path, + index_start + source_offset..index_start + source_offset + length, + ) + .await?; + drop(permit); + if bytes.len() as u64 != length { + return Err(LtxError::ChecksumMismatch); + } + hasher.update(&bytes); + let output_offset = output_start + .checked_add(source_offset) + .ok_or(LtxError::Limit(crate::LimitKind::CompactionIndexSpool))?; + let returned = replica + .host + .run(move || { + for entry in bytes.as_chunks::<{ crate::paged::ENTRY_BYTES }>().0 { + validator.validate(crate::paged::decode_index_entry(entry)?)?; } - let length = descriptor.index_length; - Ok(SpoolInput { - descriptor, - start: output_start, - length, - }) - } - .map(move |result| (order, result)) - }, - )) - .buffer_unordered(SEGMENT_TRANSFER_CONCURRENCY) - .collect::>() - .await; - let inputs = ordered_results(results)?; - sync_spool(scratch, destination, total_bytes).await?; - Ok(inputs) + file.write_all_at(output_offset, &bytes)?; + Ok::<_, LtxError>((file, validator)) + }) + .await??; + file = returned.0; + validator = returned.1; + source_offset += length; + } + if *hasher.finalize().as_bytes() != descriptor.index_digest { + return Err(LtxError::ChecksumMismatch); + } + let length = descriptor.index_length; + Ok(SpoolInput { + descriptor, + start: output_start, + length, + }) } // Finished transfers refill the same bounded window immediately. Restore diff --git a/crates/cellule-ltx/src/replica/directory/initial.rs b/crates/cellule-ltx/src/replica/directory/initial.rs index 578c00fc..7a7db472 100644 --- a/crates/cellule-ltx/src/replica/directory/initial.rs +++ b/crates/cellule-ltx/src/replica/directory/initial.rs @@ -135,7 +135,16 @@ pub(in crate::replica) async fn build_and_upload( if !leaf_entries.is_empty() { pending.push(encode_leaf_node(leaf_index, &leaf_entries)?); } - flush_node_uploads(replica, &mut pending, &mut nodes).await?; + // Delay the single-leaf upload until the root chooses inline or external. + let retained_leaf = if nodes.is_empty() && pending.len() == 1 { + let (node, bytes) = pending.pop().ok_or(LtxError::LTXCorrupted)?; + let digest = node.digest; + nodes.push(node); + vec![super::DirectoryObject { digest, bytes }] + } else { + flush_node_uploads(replica, &mut pending, &mut nodes).await?; + Vec::new() + }; if expected_page == u64::from(lock) { expected_page += 1; } @@ -165,7 +174,7 @@ pub(in crate::replica) async fn build_and_upload( } let root = nodes.pop().ok_or(LtxError::LTXCorrupted)?; Ok(DirectoryTree { - objects: Vec::new(), + objects: retained_leaf, root, height, }) diff --git a/crates/cellule-ltx/src/replica/directory/mod.rs b/crates/cellule-ltx/src/replica/directory/mod.rs index 0c7e4845..bc6c4387 100644 --- a/crates/cellule-ltx/src/replica/directory/mod.rs +++ b/crates/cellule-ltx/src/replica/directory/mod.rs @@ -254,6 +254,7 @@ pub(super) struct Verification<'a> { pub incarnation: &'a [u8; 16], pub page_size: u32, pub database_pages: u32, + pub inline_root: Option<&'a [u8]>, pub extents: &'a BTreeMap<[u8; 32], ObjectExtent>, pub host: &'a Host, pub origin: crate::LtxReadOrigin, @@ -328,7 +329,9 @@ pub(super) async fn reachable_digests( if max_objects.is_some_and(|limit| digests.len() == limit) { return Err(LtxError::Limit(crate::LimitKind::RootInventoryObjects)); } - digests.push(digest); + if inline_node(&verification, digest)?.is_none() { + digests.push(digest); + } if header.kind == 0 { let (aggregate, entries) = verify_leaf( &bytes, @@ -550,7 +553,20 @@ pub(super) async fn lookup_spans( Ok(spans) } +fn inline_node(verification: &Verification<'_>, digest: [u8; 32]) -> Result>> { + let Some(bytes) = verification.inline_root else { + return Ok(None); + }; + if bytes.len() > super::root::INLINE_DIRECTORY_BYTES || bytes.is_empty() { + return Err(LtxError::LTXCorrupted); + } + Ok((*blake3::hash(bytes).as_bytes() == digest).then(|| Arc::from(bytes))) +} + async fn read_node(verification: &Verification<'_>, digest: [u8; 32]) -> Result> { + if let Some(bytes) = inline_node(verification, digest)? { + return Ok(bytes); + } let path = verification.layout.incarnation_object_path( verification.cell, verification.incarnation, @@ -631,6 +647,9 @@ async fn read_node_uncached( verification: &Verification<'_>, digest: [u8; 32], ) -> Result> { + if let Some(bytes) = inline_node(verification, digest)? { + return Ok(bytes); + } let path = verification.layout.incarnation_object_path( verification.cell, verification.incarnation, diff --git a/crates/cellule-ltx/src/replica/directory/relocate.rs b/crates/cellule-ltx/src/replica/directory/relocate.rs index c31e74e3..1fa2005d 100644 --- a/crates/cellule-ltx/src/replica/directory/relocate.rs +++ b/crates/cellule-ltx/src/replica/directory/relocate.rs @@ -33,6 +33,7 @@ pub(in crate::replica) async fn run( incarnation: &replica.incarnation, page_size: graph.document.page_size, database_pages: graph.document.database_pages, + inline_root: graph.document.directory_inline.as_deref(), extents: &base_extents, host: &replica.host, origin: crate::LtxReadOrigin::Cold, diff --git a/crates/cellule-ltx/src/replica/directory/tests.rs b/crates/cellule-ltx/src/replica/directory/tests.rs index ca369b84..e8bf63d5 100644 --- a/crates/cellule-ltx/src/replica/directory/tests.rs +++ b/crates/cellule-ltx/src/replica/directory/tests.rs @@ -189,6 +189,7 @@ async fn incremental_update_rebuilds_the_last_height_two_branch() { incarnation: &incarnation, page_size, database_pages: base_pages, + inline_root: None, extents: &base_extents, host: &replica.host, origin: crate::LtxReadOrigin::Cold, @@ -206,6 +207,7 @@ async fn incremental_update_rebuilds_the_last_height_two_branch() { incarnation: &incarnation, page_size, database_pages: final_pages, + inline_root: None, extents: &final_extents, host: &replica.host, origin: crate::LtxReadOrigin::Cold, diff --git a/crates/cellule-ltx/src/replica/mod.rs b/crates/cellule-ltx/src/replica/mod.rs index 2ef2a4c3..42020d9e 100644 --- a/crates/cellule-ltx/src/replica/mod.rs +++ b/crates/cellule-ltx/src/replica/mod.rs @@ -18,6 +18,7 @@ mod coalesce; mod compaction; pub(crate) mod directory; mod merge; +mod packed; mod preparation; mod prepare; pub use preparation::{RootPreparation, RootPreparationFuture, RootPreparationMetadata}; @@ -327,6 +328,7 @@ pub struct CellPagedDatabase { replica: CellReplica, directory_digest: [u8; 32], directory_height: u32, + directory_inline: Option>, extents: Arc>, page_size: u32, database_pages: u32, @@ -424,6 +426,7 @@ impl CellPagedDatabase { incarnation: &self.replica.incarnation, page_size: self.page_size, database_pages: self.database_pages, + inline_root: self.directory_inline.as_deref(), extents: &self.extents, host: &self.replica.host, origin: crate::LtxReadOrigin::Cold, @@ -468,6 +471,7 @@ impl CellPagedDatabase { incarnation: &self.replica.incarnation, page_size: self.page_size, database_pages: self.database_pages, + inline_root: self.directory_inline.as_deref(), extents: &self.extents, host: &self.replica.host, origin, @@ -591,6 +595,7 @@ impl CellPagedDatabase { incarnation: &self.replica.incarnation, page_size: self.page_size, database_pages: self.database_pages, + inline_root: self.directory_inline.as_deref(), extents: &self.extents, host: &self.replica.host, origin, @@ -864,6 +869,7 @@ struct AppendBaseState { aggregate: directory::Aggregate, directory_digest: [u8; 32], directory_height: u32, + directory_inline: Option>, page_size: u32, database_pages: u32, inherited_segment_pages: Vec<[u8; 32]>, @@ -877,6 +883,7 @@ impl From for AppendBaseState { aggregate: graph.aggregate, directory_digest: graph.document.directory_digest, directory_height: graph.document.directory_height, + directory_inline: graph.document.directory_inline, page_size: graph.document.page_size, database_pages: graph.document.database_pages, inherited_segment_pages: graph.document.segment_pages, @@ -939,6 +946,7 @@ struct AppendInput { enum AppendBody { Native(Arc), Frozen(Bytes), + Packed(Arc), Bundle, } diff --git a/crates/cellule-ltx/src/replica/packed.rs b/crates/cellule-ltx/src/replica/packed.rs new file mode 100644 index 00000000..e826cfbb --- /dev/null +++ b/crates/cellule-ltx/src/replica/packed.rs @@ -0,0 +1,139 @@ +//! One bounded, content-addressed segment/index object. Native LTX bytes stay exact. +use bytes::{Bytes, BytesMut}; +use cellule_store::MultipartUploadSource; +use std::sync::Arc; + +use super::{ + AppendBody, CellReplica, PreparedSegment, SegmentDescriptor, upload::SINGLE_PUT_BYTES, +}; +use crate::{LtxError, Result}; + +pub(super) const HEADER_BYTES: u64 = 64; +const MAGIC: &[u8; 8] = b"CRBPACK1"; + +pub(super) async fn freeze( + replica: &CellReplica, + mut segment: PreparedSegment, +) -> Result { + let length = HEADER_BYTES + .checked_add(segment.descriptor.info.size_bytes) + .and_then(|n| n.checked_add(segment.index.len() as u64)) + .ok_or(LtxError::LTXCorrupted)?; + if length > SINGLE_PUT_BYTES || matches!(segment.body, AppendBody::Bundle) { + return Ok(segment); + } + let body = match &segment.body { + AppendBody::Native(source) => { + let source = source.clone(); + let size = segment.descriptor.info.size_bytes; + replica.host.run(move || source.read_small(size)).await?? + } + AppendBody::Frozen(bytes) => bytes.clone(), + AppendBody::Bundle | AppendBody::Packed(_) => return Err(LtxError::LTXCorrupted), + }; + if body.len() as u64 != segment.descriptor.info.size_bytes + || *blake3::hash(&body).as_bytes() != segment.descriptor.info.blake3 + || *blake3::hash(&segment.index).as_bytes() != segment.descriptor.index_digest + { + return Err(LtxError::ChecksumMismatch); + } + // Freeze before provider I/O; retained memory is bounded by the existing + // single-PUT limit and survives cancellation with its preparation owner. + let mut bytes = BytesMut::with_capacity(length as usize); + bytes.extend_from_slice(MAGIC); + bytes.extend_from_slice(&(body.len() as u64).to_be_bytes()); + bytes.extend_from_slice(&(segment.index.len() as u64).to_be_bytes()); + bytes.extend_from_slice(&[0; 8]); + bytes.extend_from_slice(&segment.descriptor.info.blake3); + bytes.extend_from_slice(&body); + bytes.extend_from_slice(&segment.index); + let bytes = bytes.freeze(); + let level = segment.descriptor.level(); + segment.descriptor = SegmentDescriptor::packed( + segment.descriptor.info, + segment.descriptor.index_digest, + segment.descriptor.index_length, + *blake3::hash(&bytes).as_bytes(), + ) + .with_level(level); + let source: Arc = match segment.body { + AppendBody::Native(source) => source, + AppendBody::Frozen(bytes) => Arc::new(super::upload::FrozenCapture(bytes)), + AppendBody::Bundle | AppendBody::Packed(_) => return Err(LtxError::LTXCorrupted), + }; + // Retain the pinned source and shared index, rather than all frozen bodies + // across a preparation cohort. Upload buffers only while holding host I/O. + segment.body = AppendBody::Packed(Arc::new(PackedCapture { + header: Bytes::copy_from_slice(&bytes[..HEADER_BYTES as usize]), + source, + body_length: segment.descriptor.info.size_bytes, + index: segment.index.clone(), + })); + Ok(segment) +} + +pub(super) fn verify(bytes: &Bytes, descriptor: &SegmentDescriptor) -> Result<()> { + let body_end = HEADER_BYTES + .checked_add(descriptor.info.size_bytes) + .ok_or(LtxError::LTXCorrupted)?; + let total = body_end + .checked_add(descriptor.index_length) + .ok_or(LtxError::LTXCorrupted)?; + if total > SINGLE_PUT_BYTES + || bytes.len() as u64 != total + || bytes.len() < HEADER_BYTES as usize + { + return Err(LtxError::LTXCorrupted); + } + if &bytes[..8] != MAGIC + || bytes[8..16] != descriptor.info.size_bytes.to_be_bytes() + || bytes[16..24] != descriptor.index_length.to_be_bytes() + || bytes[24..32] != [0; 8] + || bytes[32..64] != descriptor.info.blake3 + || *blake3::hash(bytes).as_bytes() != descriptor.object_digest() + || *blake3::hash(&bytes[HEADER_BYTES as usize..body_end as usize]).as_bytes() + != descriptor.info.blake3 + || *blake3::hash(&bytes[body_end as usize..]).as_bytes() != descriptor.index_digest + { + return Err(LtxError::ChecksumMismatch); + } + for entry in + crate::paged::validated_index_entries(&bytes[body_end as usize..], &descriptor.info)? + { + entry?; + } + Ok(()) +} + +struct PackedCapture { + header: Bytes, + source: Arc, + body_length: u64, + index: Bytes, +} + +#[async_trait::async_trait] +impl MultipartUploadSource for PackedCapture { + async fn byte_len(&self) -> cellule_store::Result { + // Propagate source length changes to the existing exact-upload guard. + Ok(HEADER_BYTES + self.source.byte_len().await? + self.index.len() as u64) + } + + async fn read_exact(&self, offset: u64, length: usize) -> cellule_store::Result { + let total = HEADER_BYTES + self.body_length + self.index.len() as u64; + if offset != 0 || length as u64 != total { + return Err(cellule_store::StorageError::ReadRejected { + source: Box::new(std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "packed object requires one bounded exact read", + )), + }); + } + let body = self.source.read_exact(0, self.body_length as usize).await?; + let mut bytes = BytesMut::with_capacity(length); + bytes.extend_from_slice(&self.header); + bytes.extend_from_slice(&body); + bytes.extend_from_slice(&self.index); + Ok(bytes.freeze()) + } +} diff --git a/crates/cellule-ltx/src/replica/prepare.rs b/crates/cellule-ltx/src/replica/prepare.rs index 61f98b6c..5bdb394f 100644 --- a/crates/cellule-ltx/src/replica/prepare.rs +++ b/crates/cellule-ltx/src/replica/prepare.rs @@ -604,6 +604,14 @@ impl CellReplica { self.validate_chain(&descriptors, target)?; if bundle.is_none() { prepared = coalesce::run(self, prepared).await?; + prepared = stream::iter( + prepared + .into_iter() + .map(|segment| packed::freeze(self, segment)), + ) + .buffered(SEGMENT_TRANSFER_CONCURRENCY) + .try_collect() + .await?; descriptors.truncate( base_graph .as_ref() @@ -678,6 +686,7 @@ impl CellReplica { incarnation: &self.incarnation, page_size: graph.page_size, database_pages: graph.database_pages, + inline_root: graph.directory_inline.as_deref(), extents: &base_extents, host: &self.host, origin: crate::LtxReadOrigin::Cold, @@ -695,6 +704,7 @@ impl CellReplica { incarnation: &self.incarnation, page_size, database_pages, + inline_root: None, extents: &extents, host: &self.host, origin: crate::LtxReadOrigin::Cold, @@ -717,15 +727,7 @@ impl CellReplica { } directory }; - let directory_uploads = self.put_objects( - CellObjectKind::Directory, - directory - .objects() - .iter() - .map(|node| (node.digest, node.bytes.clone())) - .collect(), - ); - let root_uploads = self.finish_root( + self.finish_root( base, base_graph .as_ref() @@ -738,11 +740,8 @@ impl CellReplica { page_size, database_pages, directory, - ); - // Both object sets are immutable; no proposal escapes unless every - // upload succeeds, and a failed sibling leaves only unreachable data. - let (_, prepared) = futures_util::future::try_join(directory_uploads, root_uploads).await?; - Ok(prepared) + ) + .await } #[expect(clippy::too_many_arguments)] @@ -783,13 +782,14 @@ impl CellReplica { if segment_pages.len() > MAX_SEGMENT_PAGES { return Err(LtxError::Limit(crate::LimitKind::CellRootSegmentPages)); } - let document = RootDocument { + let mut document = RootDocument { cell: self.cell, checksum: target.checksum, commit_sequence, database_pages, directory_digest: directory.root_digest(), directory_height: directory.height(), + directory_inline: None, incarnation: self.incarnation, page_size, schema, @@ -797,6 +797,31 @@ impl CellReplica { segments: descriptors[external_count..].to_vec(), txid: target.txid, }; + // Inline only one small leaf authenticated by this root. Large trees + // keep streamed directory objects and never grow an unbounded root. + if directory.height() == 0 + && let Some(leaf) = directory + .objects() + .iter() + .find(|node| node.digest == directory.root_digest()) + && leaf.bytes.len() <= root::INLINE_DIRECTORY_BYTES + { + document.directory_inline = Some(leaf.bytes.clone().into()); + if matches!( + encode_root(&document), + Err(LtxError::Limit(crate::LimitKind::CellRootBytes)) + ) { + document.directory_inline = None; + } + } + let directory_objects = directory + .objects() + .iter() + .filter(|node| { + document.directory_inline.is_none() || node.digest != document.directory_digest + }) + .map(|node| (node.digest, node.bytes.clone())) + .collect(); let bytes = encode_root(&document)?; let digest = *blake3::hash(&bytes).as_bytes(); root_objects.push((digest, bytes)); @@ -835,8 +860,9 @@ impl CellReplica { }; // Independent immutable objects and derivation metadata can overlap. // Neither the complete proposal nor authority rights escape on failure. - futures_util::future::try_join( + futures_util::future::try_join3( self.put_objects(CellObjectKind::Root, root_objects), + self.put_objects(CellObjectKind::Directory, directory_objects), metadata, ) .await?; diff --git a/crates/cellule-ltx/src/replica/root.rs b/crates/cellule-ltx/src/replica/root.rs index ed1c2658..2f1018e8 100644 --- a/crates/cellule-ltx/src/replica/root.rs +++ b/crates/cellule-ltx/src/replica/root.rs @@ -1,4 +1,5 @@ use serde::{Deserialize, Serialize}; +use std::sync::Arc; use crate::{Limits, LtxError, Result, SegmentInfo}; @@ -14,6 +15,7 @@ pub(super) struct RootDocument { pub database_pages: u32, pub directory_digest: [u8; 32], pub directory_height: u32, + pub directory_inline: Option>, pub incarnation: [u8; 16], pub page_size: u32, pub schema: u32, @@ -31,6 +33,7 @@ pub(super) struct SegmentDescriptor { offset: u64, length: u64, level: u8, + packed: bool, } impl SegmentDescriptor { @@ -45,6 +48,7 @@ impl SegmentDescriptor { offset: 0, length, level: 0, + packed: false, } } @@ -64,6 +68,38 @@ impl SegmentDescriptor { offset, length, level: 0, + packed: false, + } + } + + pub(super) fn packed( + info: SegmentInfo, + index_digest: [u8; 32], + index_length: u64, + object_digest: [u8; 32], + ) -> Self { + let length = info.size_bytes; + Self { + info, + index_digest, + index_length, + object_digest, + offset: super::packed::HEADER_BYTES, + length, + level: 0, + packed: true, + } + } + + pub(super) fn index_extent(&self) -> ([u8; 32], crate::CellObjectKind, u64) { + if self.packed { + ( + self.object_digest, + crate::CellObjectKind::Packed, + self.offset + self.length, + ) + } else { + (self.index_digest, crate::CellObjectKind::Index, 0) } } @@ -90,7 +126,9 @@ impl SegmentDescriptor { } pub(super) fn object_kind(&self) -> crate::CellObjectKind { - if self.object_digest == self.info.blake3 { + if self.packed { + crate::CellObjectKind::Packed + } else if self.object_digest == self.info.blake3 { crate::CellObjectKind::Ltx } else { crate::CellObjectKind::Bundle @@ -120,6 +158,13 @@ impl SegmentDescriptor { || self.length != info.size_bytes || self.offset.checked_add(self.length).is_none() || (self.object_kind() == crate::CellObjectKind::Ltx && self.offset != 0) + || (self.packed + && (self.offset != super::packed::HEADER_BYTES + || self + .offset + .checked_add(self.length) + .and_then(|n| n.checked_add(self.index_length)) + .is_none_or(|end| end > super::upload::SINGLE_PUT_BYTES))) || self .offset .checked_add(self.length) @@ -150,6 +195,7 @@ struct RootWire { database_pages: u32, directory_digest: String, directory_height: u32, + directory_inline: Option, incarnation: String, page_size: u32, schema: u32, @@ -172,13 +218,39 @@ struct SegmentWire { min_txid: String, object_digest: String, offset: String, + packed: bool, page_size: u32, post_checksum: String, pre_checksum: String, size_bytes: String, } +pub(super) const INLINE_DIRECTORY_BYTES: usize = 2 << 10; + +fn parse_inline(value: &str) -> Result> { + if value.is_empty() + || !value.len().is_multiple_of(2) + || value.len() > INLINE_DIRECTORY_BYTES * 2 + { + return Err(LtxError::LTXCorrupted); + } + value + .as_bytes() + .chunks_exact(2) + .map(|pair| Ok((nibble(pair[0])? << 4) | nibble(pair[1])?)) + .collect::>>() + .map(Into::into) +} + pub(super) fn encode_root(root: &RootDocument) -> Result> { + if let Some(bytes) = &root.directory_inline + && (root.directory_height != 0 + || bytes.is_empty() + || bytes.len() > INLINE_DIRECTORY_BYTES + || *blake3::hash(bytes).as_bytes() != root.directory_digest) + { + return Err(LtxError::LTXCorrupted); + } if root.schema == 0 || root.txid == 0 || root.checksum & crate::CHECKSUM_FLAG == 0 @@ -197,6 +269,10 @@ pub(super) fn encode_root(root: &RootDocument) -> Result> { database_pages: root.database_pages, directory_digest: encode_hex(&root.directory_digest), directory_height: root.directory_height, + directory_inline: root + .directory_inline + .as_ref() + .map(|bytes| encode_hex(bytes)), incarnation: encode_hex(&root.incarnation), page_size: root.page_size, schema: root.schema, @@ -207,7 +283,7 @@ pub(super) fn encode_root(root: &RootDocument) -> Result> { .collect(), segments: root.segments.iter().map(segment_wire).collect(), txid: root.txid.to_string(), - version: 1, + version: 2, })?; if bytes.len() as u64 > ROOT_BYTES { return Err(LtxError::Limit(crate::LimitKind::CellRootBytes)); @@ -230,7 +306,7 @@ pub(super) fn decode_root(bytes: &[u8]) -> Result { return Err(LtxError::Limit(crate::LimitKind::CellRootBytes)); } let wire: RootWire = serde_json::from_slice(bytes)?; - if wire.version != 1 { + if wire.version != 2 { return Err(LtxError::LTXCorrupted); } let root = RootDocument { @@ -240,6 +316,10 @@ pub(super) fn decode_root(bytes: &[u8]) -> Result { database_pages: wire.database_pages, directory_digest: parse_hex(&wire.directory_digest)?, directory_height: wire.directory_height, + directory_inline: wire + .directory_inline + .map(|value| parse_inline(&value)) + .transpose()?, incarnation: parse_hex(&wire.incarnation)?, page_size: wire.page_size, schema: wire.schema, @@ -296,6 +376,7 @@ fn segment_wire(segment: &SegmentDescriptor) -> SegmentWire { max_txid: segment.info.max_txid.to_string(), min_txid: segment.info.min_txid.to_string(), object_digest: encode_hex(&segment.object_digest), + packed: segment.packed, offset: segment.offset.to_string(), page_size: segment.info.page_size, post_checksum: checksum(segment.info.post_checksum), @@ -322,6 +403,7 @@ fn segment_descriptor(segment: SegmentWire) -> Result { offset: decimal(&segment.offset)?, length: decimal(&segment.length)?, level: segment.level, + packed: segment.packed, }) } @@ -383,6 +465,7 @@ mod tests { database_pages: 9, directory_digest: [2; 32], directory_height: 0, + directory_inline: None, incarnation: [3; 16], page_size: 4096, schema: 1, diff --git a/crates/cellule-ltx/src/replica/upload.rs b/crates/cellule-ltx/src/replica/upload.rs index 09b758a1..ebf383ca 100644 --- a/crates/cellule-ltx/src/replica/upload.rs +++ b/crates/cellule-ltx/src/replica/upload.rs @@ -67,6 +67,20 @@ impl CellReplica { index, body, } = segment; + if let AppendBody::Packed(source) = body { + let path = self.layout.incarnation_object_path( + &self.cell, + &self.incarnation, + &descriptor.object_digest(), + CellObjectKind::Packed, + ); + let _permit = self.host.io_permit().await?; + let length = + packed::HEADER_BYTES + descriptor.info.size_bytes + descriptor.index_length; + put_source(self, &path, source, length, descriptor.object_digest()).await?; + self.cost.record(length); + return Ok(()); + } let body_upload = async { if descriptor.object_kind() != CellObjectKind::Ltx { return Ok(()); @@ -74,7 +88,7 @@ impl CellReplica { let source: Arc = match body { AppendBody::Native(source) => source, AppendBody::Frozen(bytes) => Arc::new(FrozenCapture(bytes)), - AppendBody::Bundle => { + AppendBody::Bundle | AppendBody::Packed(_) => { return Err(LtxError::InvalidState("native Cell body source missing")); } }; @@ -137,7 +151,7 @@ const MULTIPART_BYTES: usize = 8 << 20; // below one multipart chunk so many publishing Cells stay within node memory. pub(super) const SINGLE_PUT_BYTES: u64 = 256 << 10; -struct FrozenCapture(Bytes); +pub(super) struct FrozenCapture(pub(super) Bytes); #[async_trait::async_trait] impl cellule_store::MultipartUploadSource for FrozenCapture { diff --git a/crates/cellule-ltx/src/replica/verify.rs b/crates/cellule-ltx/src/replica/verify.rs index 85ef66c3..295e295d 100644 --- a/crates/cellule-ltx/src/replica/verify.rs +++ b/crates/cellule-ltx/src/replica/verify.rs @@ -57,6 +57,7 @@ impl CellReplica { incarnation: &self.incarnation, page_size: graph.document.page_size, database_pages: graph.document.database_pages, + inline_root: graph.document.directory_inline.as_deref(), extents: &extents, host: &self.host, origin: crate::LtxReadOrigin::Cold, @@ -87,6 +88,32 @@ impl CellReplica { }), ); for descriptor in &graph.descriptors { + if descriptor.object_kind() == CellObjectKind::Packed { + let path = self.layout.incarnation_object_path( + &self.cell, + &self.incarnation, + &descriptor.object_digest(), + CellObjectKind::Packed, + ); + let _permit = self.host.io_permit().await?; + let result = self + .layout + .store() + .get_with_etag_bounded(&path, upload::SINGLE_PUT_BYTES) + .await; + self.host.observe_ltx_origin_request( + crate::LtxReadOrigin::Cold, + result.is_ok(), + result.as_ref().map_or(0, |(bytes, _)| bytes.len()), + ); + let (bytes, _) = result?; + packed::verify(&bytes, descriptor)?; + objects.insert(RootObjectRef { + digest: descriptor.object_digest(), + kind: CellObjectKind::Packed, + }); + continue; + } let body = RootObjectRef { digest: descriptor.object_digest(), kind: descriptor.object_kind(), @@ -274,6 +301,7 @@ impl CellReplica { incarnation: &self.incarnation, page_size: document.page_size, database_pages: document.database_pages, + inline_root: document.directory_inline.as_deref(), extents: &extents, host: &self.host, origin: crate::LtxReadOrigin::Cold, @@ -435,6 +463,7 @@ impl VerifiedRoot { replica, directory_digest: document.directory_digest, directory_height: document.directory_height, + directory_inline: document.directory_inline.clone(), extents: Arc::new(extents), page_size: document.page_size, database_pages: document.database_pages, diff --git a/crates/cellule-ltx/tests/cell/restore.rs b/crates/cellule-ltx/tests/cell/restore.rs index 751f44a8..978d8bc1 100644 --- a/crates/cellule-ltx/tests/cell/restore.rs +++ b/crates/cellule-ltx/tests/cell/restore.rs @@ -196,7 +196,7 @@ impl ObjectStore for InstrumentedStore { self.puts.fetch_add(1, Ordering::SeqCst); self.check_upload_fault()?; let result = self.inner.put_opts(location, payload, options).await?; - if location.extension() == Some("ltx") + if matches!(location.extension(), Some("ltx" | "pack")) && self .fault .compare_exchange( diff --git a/crates/cellule-ltx/tests/cell/roots.rs b/crates/cellule-ltx/tests/cell/roots.rs index 85080c9a..a27fcf23 100644 --- a/crates/cellule-ltx/tests/cell/roots.rs +++ b/crates/cellule-ltx/tests/cell/roots.rs @@ -37,6 +37,7 @@ mod compaction; mod compaction_transfers; mod directory; mod lifecycle; +mod packed; mod preparation; mod prepare_cost; mod sparse; diff --git a/crates/cellule-ltx/tests/cell/roots/coalescing.rs b/crates/cellule-ltx/tests/cell/roots/coalescing.rs index a6833c4d..a4f70cae 100644 --- a/crates/cellule-ltx/tests/cell/roots/coalescing.rs +++ b/crates/cellule-ltx/tests/cell/roots/coalescing.rs @@ -44,8 +44,8 @@ async fn small_captured_batch_publishes_one_delta_and_restores_every_original_cu ); assert_eq!( counted.put_requests(), - 4, - "one body/index pair, final directory and root" + 2, + "one packed body/index and root containing its final directory" ); let retried = cell.prepare(Some(&base), &batch, 17, 1).await.unwrap(); assert_eq!(retried.root(), prepared.root()); diff --git a/crates/cellule-ltx/tests/cell/roots/compaction.rs b/crates/cellule-ltx/tests/cell/roots/compaction.rs index ce54e367..857b2664 100644 --- a/crates/cellule-ltx/tests/cell/roots/compaction.rs +++ b/crates/cellule-ltx/tests/cell/roots/compaction.rs @@ -3,6 +3,66 @@ use super::*; use cellule_ltx::LtxError; +#[tokio::test] +async fn small_compaction_fetches_each_verified_pack_once_for_body_and_index() { + let directory = tempfile::tempdir().unwrap(); + let counted = Arc::new(cellule_store::test_support::CountingObjectStore::new( + Arc::new(InMemory::new()), + )); + let cell = replica(Store::new(counted.clone()), [225; 32], [226; 16]); + let mut writer = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); + let mut root = None; + for sequence in 1..=4 { + writer + .transaction(|tx| { + if sequence == 1 { + tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(0)")?; + } + tx.execute("UPDATE t SET v=?1", [sequence])?; + Ok(()) + }) + .unwrap(); + root = Some( + cell.prepare(root.as_ref(), &writer.capture().unwrap(), sequence, 1) + .await + .unwrap() + .root(), + ); + } + let root = root.unwrap(); + counted.reset(); + let compacted = cell + .prepare_compaction(&root, 0..4, 1, directory.path()) + .await + .unwrap(); + let packs = counted + .requests() + .into_iter() + .filter(|request| request.location.ends_with(".pack")) + .collect::>(); + assert_eq!( + packs.len(), + 4, + "index and body must share one verified origin fetch" + ); + assert!( + packs + .iter() + .all(|request| request.kind == cellule_store::test_support::ObjectReadKind::Full) + ); + assert_eq!(compacted.root().position, root.position); + assert_eq!(compacted.root().commit_sequence, root.commit_sequence); + let restored = directory.path().join("restored"); + compacted.verified().restore(&restored).await.unwrap(); + let db = cellule_ltx::rusqlite::Connection::open(restored).unwrap(); + assert_eq!( + db.query_row("SELECT v FROM t", [], |row| row.get::<_, i64>(0)) + .unwrap(), + 4 + ); + writer.close().unwrap(); +} + #[tokio::test] async fn private_compaction_append_retains_original_predecessor_and_exact_root() { let directory = tempfile::tempdir().unwrap(); @@ -101,7 +161,6 @@ async fn scheduled_compaction_append_checks_origin_and_sources_before_escape() { let incarnation = [212; 16]; let replica = CellReplica::new(layout.clone(), cell, incarnation, Limits::default()).unwrap(); let mut root = None; - let mut first_body = None; for sequence in 1..=8 { writer .transaction(|transaction| { @@ -114,9 +173,6 @@ async fn scheduled_compaction_append_checks_origin_and_sources_before_escape() { }) .unwrap(); let cuts = writer.capture().unwrap(); - if sequence == 1 { - first_body = Some(cuts.segments[0].info().blake3); - } root = Some( replica .prepare(root.as_ref(), &cuts, sequence, 1) @@ -211,11 +267,15 @@ async fn scheduled_compaction_append_checks_origin_and_sources_before_escape() { ); backend.put(&path, bytes.into()).await.unwrap(); + let selected_body = objects + .iter() + .find(|object| object.kind == CellObjectKind::Packed) + .unwrap(); let body_path = layout.incarnation_object_path( &cell, &incarnation, - &first_body.unwrap(), - CellObjectKind::Ltx, + &selected_body.digest, + selected_body.kind, ); let original = backend .get(&body_path) @@ -255,8 +315,8 @@ async fn scheduled_compaction_append_checks_origin_and_sources_before_escape() { assert_eq!(prepared.verified().schema(), 2); assert_eq!( counted.put_requests(), - 6, - "one-leaf compaction append needs two body/index pairs, final directory and root" + 3, + "one-leaf compaction append needs two packed segments and its root with an inline directory" ); let compacted = replica .prepare_scheduled_compaction(&root, scratch.path()) @@ -932,9 +992,6 @@ async fn compaction_rejects_selected_body_corruption_outside_page_frames() { }) .unwrap(); let batch = writer.capture().unwrap(); - let info = batch.segments[0].info().clone(); - let mut corrupted = std::fs::read(batch.segments[0].path()).unwrap(); - corrupted[0] ^= 1; let inner = Arc::new(InMemory::new()); let store = Store::new(inner.clone()); let cell = [93; 32]; @@ -943,8 +1000,23 @@ async fn compaction_rejects_selected_body_corruption_outside_page_frames() { let replica = CellReplica::new(layout.clone(), cell, incarnation, Limits::default()).unwrap(); let root = replica.prepare(None, &batch, 1, 1).await.unwrap().root(); writer.close().unwrap(); - let object = - layout.incarnation_object_path(&cell, &incarnation, &info.blake3, CellObjectKind::Ltx); + let body = replica + .reachable_objects(&root) + .await + .unwrap() + .into_iter() + .find(|object| object.kind == CellObjectKind::Packed) + .unwrap(); + let object = layout.incarnation_object_path(&cell, &incarnation, &body.digest, body.kind); + let mut corrupted = inner + .get(&object) + .await + .unwrap() + .bytes() + .await + .unwrap() + .to_vec(); + corrupted[64] ^= 1; // Native LTX header, outside every authenticated page frame. inner .put(&object, Bytes::from(corrupted).into()) .await diff --git a/crates/cellule-ltx/tests/cell/roots/compaction_transfers.rs b/crates/cellule-ltx/tests/cell/roots/compaction_transfers.rs index e5c1128b..52d970e9 100644 --- a/crates/cellule-ltx/tests/cell/roots/compaction_transfers.rs +++ b/crates/cellule-ltx/tests/cell/roots/compaction_transfers.rs @@ -77,7 +77,10 @@ impl ObjectStore for TransferGateStore { } async fn get_opts(&self, path: &Path, opts: GetOptions) -> object_store::Result { let kind = *self.kind.lock().unwrap(); - let selected = opts.range.is_some() && kind.is_some() && path.extension() == kind; + let selected = !opts.head + && (opts.range.is_some() || kind == Some("pack")) + && kind.is_some() + && path.extension() == kind; let _active = selected.then(|| { let active = self.active.fetch_add(1, Ordering::AcqRel) + 1; self.peak.fetch_max(active, Ordering::AcqRel); @@ -130,14 +133,22 @@ async fn compaction_index_transfers_refill_before_first_source_finishes() { verify_transfer_refill("index").await; } +#[tokio::test] +async fn packed_compaction_transfers_refill_before_first_source_finishes() { + verify_transfer_refill("pack").await; +} + async fn verify_transfer_refill(kind: &'static str) { let directory = tempfile::tempdir().unwrap(); let backend = Arc::new(TransferGateStore::new()); let cell = replica(Store::new(backend.clone()), [233; 32], [234; 16]); let mut writer = Db::open(&directory.path().join("writer.sqlite"), Limits::default()).unwrap(); + let padding = if kind == "pack" { 96 } else { 300_000 }; writer .transaction(|tx| { - tx.execute_batch("CREATE TABLE counter(value); INSERT INTO counter VALUES(0)") + tx.execute_batch("CREATE TABLE counter(value); INSERT INTO counter VALUES(0); CREATE TABLE padding(v)")?; + tx.execute("INSERT INTO padding VALUES(randomblob(?1))", [padding])?; + Ok(()) }) .unwrap(); let mut root = cell @@ -147,7 +158,11 @@ async fn verify_transfer_refill(kind: &'static str) { .root(); for sequence in 2..=5 { writer - .transaction(|tx| tx.execute_batch("UPDATE counter SET value=value+1")) + .transaction(|tx| { + tx.execute_batch("UPDATE counter SET value=value+1")?; + tx.execute("UPDATE padding SET v=randomblob(?1)", [padding])?; + Ok(()) + }) .unwrap(); root = cell .prepare(Some(&root), &writer.capture().unwrap(), sequence, 1) diff --git a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs index fe2a101a..64a23f65 100644 --- a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs +++ b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs @@ -124,23 +124,18 @@ async fn small_appends_report_a_bounded_publication_cost() { ) .unwrap(); - // A bootstrap root uploads one body, one index, the changed directory - // nodes, and the root document. The bound documents the per-command - // object-store amplification the runtime budgets against. + // A small root uses one packed body/index and an authenticated inline leaf. + // Runtime lineage and control selection add two authority PUTs. let first = writer.capture().unwrap(); let root = replica.prepare(None, &first, 1, 1).await.unwrap().root(); let initial = replica.take_publication_cost(); - assert!( - (4..=12).contains(&initial.objects), - "bootstrap objects: {initial:?}" - ); + assert_eq!(initial.objects, 2, "bootstrap objects: {initial:?}"); assert!( initial.bytes >= first.segments[0].info().size_bytes, "{initial:?}" ); - // One appended command pays at least a body and index, and no more than the - // same bounded set of metadata objects. + // The appended command retains the same bounded representation. writer .transaction(|transaction| { transaction.execute_batch("INSERT INTO t VALUES(randomblob(4096))") @@ -149,10 +144,7 @@ async fn small_appends_report_a_bounded_publication_cost() { let second = writer.capture().unwrap(); replica.prepare(Some(&root), &second, 2, 1).await.unwrap(); let append = replica.take_publication_cost(); - assert!( - (3..=12).contains(&append.objects), - "append objects: {append:?}" - ); + assert_eq!(append.objects, 2, "append objects: {append:?}"); assert!( append.bytes >= second.segments[0].info().size_bytes, "{append:?}" diff --git a/crates/cellule-ltx/tests/cell/roots/packed.rs b/crates/cellule-ltx/tests/cell/roots/packed.rs new file mode 100644 index 00000000..0a1fb16b --- /dev/null +++ b/crates/cellule-ltx/tests/cell/roots/packed.rs @@ -0,0 +1,145 @@ +//! Packed origin bytes, extents and inline directory corruption fail closed. +use super::*; + +#[tokio::test] +async fn packed_root_vector_authenticates_exact_native_bytes_and_every_metadata_extent() { + let directory = tempfile::tempdir().unwrap(); + let mut writer = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); + writer + .transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let cuts = writer.capture().unwrap(); + let native = std::fs::read(cuts.segments[0].path()).unwrap(); + let backend = Arc::new(InMemory::new()); + let store = Store::new(backend.clone()); + let layout = CellStorageLayout::new(store.clone(), Path::from("runtime"), [3; 16]); + let cell = [241; 32]; + let incarnation = [242; 16]; + let replica = replica(store, cell, incarnation); + let prepared = replica.prepare(None, &cuts, 1, 1).await.unwrap(); + let root = prepared.root(); + let root_path = + layout.incarnation_object_path(&cell, &incarnation, &root.digest, CellObjectKind::Root); + let root_bytes = backend + .get(&root_path) + .await + .unwrap() + .bytes() + .await + .unwrap(); + let wire: serde_json::Value = serde_json::from_slice(&root_bytes).unwrap(); + assert_eq!(wire["version"], 2); + assert_eq!(wire["directory_height"], 0); + assert!(wire["directory_inline"].as_str().unwrap().len() <= 4096); + assert_eq!(wire["segments"][0]["packed"], true); + assert_eq!(wire["segments"][0]["offset"], "64"); + let objects = replica.reachable_objects(&root).await.unwrap(); + assert_eq!(objects.len(), 2); + let packed = objects + .iter() + .find(|object| object.kind == CellObjectKind::Packed) + .unwrap(); + let path = layout.incarnation_object_path(&cell, &incarnation, &packed.digest, packed.kind); + let bytes = backend.get(&path).await.unwrap().bytes().await.unwrap(); + assert_eq!(&bytes[..8], b"CRBPACK1"); + assert_eq!(&bytes[8..16], &(native.len() as u64).to_be_bytes()); + assert_eq!(&bytes[24..32], &[0; 8]); + assert_eq!(&bytes[32..64], &cuts.segments[0].info().blake3); + assert_eq!(&bytes[64..64 + native.len()], native.as_slice()); + assert_eq!(*blake3::hash(&bytes).as_bytes(), packed.digest); + writer.close().unwrap(); + let expected = directory.path().join("expected"); + restore_exact( + &VerifiedPlan::new(&cuts.segments, cuts.position, Limits::default()).unwrap(), + &expected, + ) + .unwrap(); + + // Header, native body and index all belong to the selected packed object. + for position in [0, 8, 16, 24, 32, 64, bytes.len() - 1] { + let mut corrupt = bytes.to_vec(); + corrupt[position] ^= 1; + backend + .put(&path, Bytes::from(corrupt).into()) + .await + .unwrap(); + assert!( + replica.reachable_objects(&root).await.is_err(), + "byte {position}" + ); + assert!( + replica + .prepare_compaction(&root, 0..1, 9, directory.path()) + .await + .is_err(), + "byte {position}" + ); + } + backend + .put(&path, bytes.slice(..bytes.len() - 1).into()) + .await + .unwrap(); + assert!(replica.reachable_objects(&root).await.is_err()); + backend.delete(&path).await.unwrap(); + assert!(replica.reachable_objects(&root).await.is_err()); + backend.put(&path, bytes.into()).await.unwrap(); + let restored = directory.path().join("restored"); + prepared.verified().restore(&restored).await.unwrap(); + assert_eq!( + std::fs::read(&restored).unwrap(), + std::fs::read(expected).unwrap() + ); +} + +#[tokio::test] +async fn packed_root_rejects_cross_scope_and_noncanonical_overlapping_extents() { + let directory = tempfile::tempdir().unwrap(); + let mut writer = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); + writer + .transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) + .unwrap(); + let cuts = writer.capture().unwrap(); + let backend = Arc::new(InMemory::new()); + let store = Store::new(backend.clone()); + let layout = CellStorageLayout::new(store.clone(), Path::from("runtime"), [3; 16]); + let cell = [243; 32]; + let incarnation = [244; 16]; + let replica = replica(store.clone(), cell, incarnation); + let root = replica.prepare(None, &cuts, 1, 1).await.unwrap().root(); + let path = + layout.incarnation_object_path(&cell, &incarnation, &root.digest, CellObjectKind::Root); + let bytes = backend.get(&path).await.unwrap().bytes().await.unwrap(); + let original: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + for (field, value) in [ + ("offset", serde_json::json!("0")), + ("offset", serde_json::json!("65")), + ("index_length", serde_json::json!("18446744073709551615")), + ("length", serde_json::json!("18446744073709551615")), + ] { + let mut wire = original.clone(); + wire["segments"][0][field] = value; + let bytes = serde_json::to_vec(&wire).unwrap(); + let bad = RootRef { + digest: *blake3::hash(&bytes).as_bytes(), + ..root + }; + let path = + layout.incarnation_object_path(&cell, &incarnation, &bad.digest, CellObjectKind::Root); + backend.put(&path, Bytes::from(bytes).into()).await.unwrap(); + assert!(replica.open_root(&bad).await.is_err(), "{field}"); + } + let foreign = super::replica(store, [245; 32], incarnation); + assert!(foreign.open_root(&root).await.is_err()); + let mut wire = original; + wire["directory_inline"] = serde_json::json!("00".repeat(2049)); + let bytes = serde_json::to_vec(&wire).unwrap(); + let bad = RootRef { + digest: *blake3::hash(&bytes).as_bytes(), + ..root + }; + let path = + layout.incarnation_object_path(&cell, &incarnation, &bad.digest, CellObjectKind::Root); + backend.put(&path, Bytes::from(bytes).into()).await.unwrap(); + assert!(replica.open_root(&bad).await.is_err()); + writer.close().unwrap(); +} diff --git a/crates/cellule-ltx/tests/cell/roots/prepare_cost.rs b/crates/cellule-ltx/tests/cell/roots/prepare_cost.rs index 11b98cd2..c8f5fdca 100644 --- a/crates/cellule-ltx/tests/cell/roots/prepare_cost.rs +++ b/crates/cellule-ltx/tests/cell/roots/prepare_cost.rs @@ -60,8 +60,8 @@ async fn small_roots_keep_descriptor_metadata_inline_and_check_origin() { .root(); assert_eq!( counted.put_requests(), - 4, - "small roots upload body, index, directory and root without a separate descriptor page" + 2, + "small roots upload one authenticated packed segment and the root containing its directory leaf" ); assert_eq!(counted.counts().full, 0); assert_eq!( @@ -69,7 +69,7 @@ async fn small_roots_keep_descriptor_metadata_inline_and_check_origin() { usize::from(root.is_some()), "the authenticated predecessor root must still exist at origin" ); - assert_eq!(replica.take_publication_cost().objects, 4); + assert_eq!(replica.take_publication_cost().objects, 2); writer.prune_captured(&cuts).unwrap(); root = Some(next); } @@ -159,8 +159,8 @@ async fn append_reuses_verified_descriptor_pages_and_rejects_missing_origin() { .root(); assert_eq!( counted.put_requests(), - 4, - "only new body, index, directory and root with an inline tail" + 3, + "only new packed segment, external directory and root with an inline tail" ); assert_eq!( counted.counts().full, @@ -172,7 +172,7 @@ async fn append_reuses_verified_descriptor_pages_and_rejects_missing_origin() { 3, "predecessor root and both descriptor pages still checked" ); - assert_eq!(replica.take_publication_cost().objects, 4); + assert_eq!(replica.take_publication_cost().objects, 3); writer.prune_captured(&next).unwrap(); let restored = directory.path().join("restored.sqlite"); diff --git a/crates/cellule-ltx/tests/host/hooks/compaction.rs b/crates/cellule-ltx/tests/host/hooks/compaction.rs index c48cd566..1daf4073 100644 --- a/crates/cellule-ltx/tests/host/hooks/compaction.rs +++ b/crates/cellule-ltx/tests/host/hooks/compaction.rs @@ -59,6 +59,7 @@ async fn compaction_case_inputs( replica: &CellReplica, writer: &mut Db, composed: bool, + large: bool, ) -> (cellule_ltx::RootRef, Option) { let mut root = replica .prepare(None, &writer.capture().unwrap(), 1, 1) @@ -70,7 +71,13 @@ async fn compaction_case_inputs( } for sequence in 2..=8 { writer - .transaction(|tx| tx.execute_batch("UPDATE t SET v=randomblob(20000)")) + .transaction(|tx| { + tx.execute_batch("UPDATE t SET v=randomblob(20000)")?; + if large { + tx.execute_batch("UPDATE padding SET v=randomblob(400000)")?; + } + Ok(()) + }) .unwrap(); root = replica .prepare(Some(&root), &writer.capture().unwrap(), sequence, 1) @@ -199,10 +206,14 @@ async fn canceled_parallel_output_flushes_retain_both_jobs_and_scratch() { async fn compaction_dependency_failure_cannot_return_metadata_proposal() { use cellule_store::test_support::CountingObjectStore; - for (composed, operation) in [false, true].into_iter().flat_map(|composed| { + for (composed, operation, packed) in [false, true].into_iter().flat_map(|composed| { ["compaction_ltx_read", "compaction_index_read"] .into_iter() - .map(move |operation| (composed, operation)) + .flat_map(move |operation| { + [false, true] + .into_iter() + .map(move |packed| (composed, operation, packed)) + }) }) { let (directory, faults, host, mut writer) = fixture(); let counted = Arc::new(CountingObjectStore::new(Arc::new(InMemory::new()))); @@ -218,7 +229,16 @@ async fn compaction_dependency_failure_cannot_return_metadata_proposal() { ) .unwrap() .with_host(host); - let (root, cuts) = compaction_case_inputs(&replica, &mut writer, composed).await; + if !packed { + writer + .transaction(|tx| { + tx.execute_batch( + "CREATE TABLE padding(v); INSERT INTO padding VALUES(randomblob(400000))", + ) + }) + .unwrap(); + } + let (root, cuts) = compaction_case_inputs(&replica, &mut writer, composed, !packed).await; counted.reset(); faults.arm(Some(operation)); let result = if let Some(cuts) = &cuts { @@ -232,10 +252,18 @@ async fn compaction_dependency_failure_cannot_return_metadata_proposal() { .await }; assert!(result.is_err(), "metadata upload must not mask {operation}"); - assert!( - counted.put_requests() > 0, - "the independent metadata branch should complete" - ); + if packed { + assert_eq!( + counted.put_requests(), + 0, + "packed identity is not known before source verification" + ); + } else { + assert!( + counted.put_requests() > 0, + "the independent metadata branch should complete: composed={composed}, operation={operation}" + ); + } faults.arm(None); let restored = directory.path().join("original.sqlite"); replica @@ -534,10 +562,9 @@ async fn compaction_overlaps_independent_remote_transfers() { .await .unwrap(); - // Cached metadata HEADs, source transfers, directory and root uploads each - // need one wave. The compacted body/index upload overlaps the directory; - // the returned proposal still waits for all immutable dependencies. - assert_eq!(started.elapsed(), delay * 4); + // Cached root HEADs, concurrent packed source transfers, then packed/root + // uploads need three waves. The bounded directory leaf stays in the root. + assert_eq!(started.elapsed(), delay * 3); writer.close().unwrap(); } #[cfg(feature = "replica")] @@ -740,7 +767,7 @@ async fn verify_canceled_compaction(pre_admitted: bool) { .with_recovery_slots(recovery.clone()) .with_scratch_slots(scratch.clone()), ); - let (root, cuts) = compaction_case_inputs(&replica, &mut writer, composed).await; + let (root, cuts) = compaction_case_inputs(&replica, &mut writer, composed, false).await; writer.close().unwrap(); let pause = Arc::new(Pause { operation, diff --git a/crates/cellule-ltx/tests/host/hooks/injection.rs b/crates/cellule-ltx/tests/host/hooks/injection.rs index fabbe8f5..7defcf06 100644 --- a/crates/cellule-ltx/tests/host/hooks/injection.rs +++ b/crates/cellule-ltx/tests/host/hooks/injection.rs @@ -47,6 +47,11 @@ async fn rustfs_directory_cache_fill_releases_origin_admission_before_local_sync async fn verify_cache_fill_admission(store: impl Fn() -> Store, prefix: &str, job_count: usize) { let (directory, faults, host, mut writer) = fixture(); + // More than 2 KiB of directory entries exercises persistent external-node + // cache filling; an inline leaf intentionally has no cache write. + writer + .transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(randomblob(100000))")) + .unwrap(); let replica = |cell| { CellReplica::new( CellStorageLayout::new(store(), ObjectPath::from(prefix), [231; 16]), @@ -141,7 +146,7 @@ async fn verify_cache_fill_admission(store: impl Fn() -> Store, prefix: &str, jo let count: u64 = db .query_row("SELECT count(*) FROM t", [], |row| row.get(0)) .unwrap(); - assert_eq!(count, 1); + assert_eq!(count, 2); } #[test] diff --git a/crates/cellule-ltx/tests/host/hooks/prepare.rs b/crates/cellule-ltx/tests/host/hooks/prepare.rs index ff780c14..9f52561b 100644 --- a/crates/cellule-ltx/tests/host/hooks/prepare.rs +++ b/crates/cellule-ltx/tests/host/hooks/prepare.rs @@ -310,7 +310,12 @@ async fn cell_prepare_bounds_source_transfers_without_local_writes() { .iter() .map(|segment| segment.info().size_bytes.div_ceil(8 << 20) as usize) .sum(); - assert_eq!(faults.read_calls.load(Ordering::Relaxed), expected_reads); + // The small bootstrap also needs one bounded read to derive the packed + // content digest before upload; its frozen upload rechecks the source. + assert_eq!( + faults.read_calls.load(Ordering::Relaxed), + expected_reads + 1 + ); writer.close().unwrap(); drop(directory); } @@ -432,7 +437,7 @@ async fn warm_append_reuses_its_authenticated_root_metadata() { assert_eq!(started.elapsed(), delay * 2); assert_eq!(counted.counts().heads, 1); assert_eq!(counted.counts().body_requests(), 0); - assert_eq!(counted.put_requests(), 4); + assert_eq!(counted.put_requests(), 2); assert_eq!(prepared.root().position, second.position); counted.reset(); let compacted = replica @@ -444,19 +449,18 @@ async fn warm_append_reuses_its_authenticated_root_metadata() { compacted.root().commit_sequence, prepared.root().commit_sequence ); - // Remote range reads belong to compaction's authenticated body/index - // spools, rather than ordinary warm-write SQLite page faults. - assert_eq!(counted.counts().ranges, 4); - for extension in [".ltx", ".index"] { + // Each compaction input supplies both authenticated scratch streams from + // one pack GET; ordinary warm append still makes no metadata body reads. + assert_eq!(counted.counts().ranges, 0); + assert_eq!(counted.counts().full, 2); + for (kind, expected) in [(ObjectReadKind::Range, 0), (ObjectReadKind::Full, 2)] { assert_eq!( counted .requests() .iter() - .filter(|request| { - request.kind == ObjectReadKind::Range && request.location.ends_with(extension) - }) + .filter(|request| { request.kind == kind && request.location.ends_with(".pack") }) .count(), - 2 + expected ); } let independent = CellReplica::new( @@ -478,7 +482,7 @@ async fn warm_append_reuses_its_authenticated_root_metadata() { .unwrap(); let cold_started = tokio::time::Instant::now(); independent.open_root(&prepared.root()).await.unwrap(); - assert!(cold_started.elapsed() >= delay * 2); + assert_eq!(cold_started.elapsed(), delay); writer.close().unwrap(); } #[cfg(feature = "replica")] diff --git a/crates/cellule-peer-http/src/performance_tests.rs b/crates/cellule-peer-http/src/performance_tests.rs index 934dc8b9..c998d9f6 100644 --- a/crates/cellule-peer-http/src/performance_tests.rs +++ b/crates/cellule-peer-http/src/performance_tests.rs @@ -291,6 +291,7 @@ fn publication_samples_cover_grouped_commands_without_inventing_events() { total: Duration::from_millis(sequence * 3), succeeded: true, commit_sequence: sequence, + covered_commits: if sequence == 1 { 1 } else { 4 }, }, ); } @@ -316,6 +317,7 @@ fn publication_samples_reject_missing_final_root() { total: Duration::ZERO, succeeded: true, commit_sequence: 1, + covered_commits: 1, }, ); publications.covered_publications(1, 5); diff --git a/crates/cellule-runtime/api-prelude.txt b/crates/cellule-runtime/api-prelude.txt index a6000af1..6c2b29c7 100644 --- a/crates/cellule-runtime/api-prelude.txt +++ b/crates/cellule-runtime/api-prelude.txt @@ -5,6 +5,7 @@ BlobNamespace BuildDescriptor CatalogRole CellClient +CellPublicationProgress CellReadReplica CellId CellModule diff --git a/crates/cellule-runtime/src/cell/actor/inventory/tests.rs b/crates/cellule-runtime/src/cell/actor/inventory/tests.rs index 3df9611f..3c0df482 100644 --- a/crates/cellule-runtime/src/cell/actor/inventory/tests.rs +++ b/crates/cellule-runtime/src/cell/actor/inventory/tests.rs @@ -103,6 +103,7 @@ async fn stale_actor_probe_cannot_clear_newer_mutation_markers_or_replace_newer_ durability_submitter: publisher.durability_submitter(), publisher: Some(publisher), publications: VecDeque::new(), + publishing_since: None, publication_bytes: 0, unpublished_node_logs: 0, queue: VecDeque::new(), diff --git a/crates/cellule-runtime/src/cell/actor/mod.rs b/crates/cellule-runtime/src/cell/actor/mod.rs index 4ed0b7a5..80ece725 100644 --- a/crates/cellule-runtime/src/cell/actor/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/mod.rs @@ -143,6 +143,21 @@ pub struct NodeJobReservation { _permit: tokio::sync::OwnedSemaphorePermit, } +/// Actor-owned committed publication debt, sampled without provider I/O. +/// +/// These observations grant no durability proof. Age covers queued and running +/// command publications, excluding migration and failed recovery obligations. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CellPublicationProgress { + /// Physical capture publications awaiting their terminal result. A capture + /// may contain several logically committed commands. + pub pending_publications: usize, + /// Retained capture bytes of those command publications. + pub retained_capture_bytes: u64, + /// Age of the oldest queued or running command publication, if any. + pub oldest_unpublished: Option, +} + /// Point-in-time node admission usage for one embedded Cell runtime. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct CellRuntimeStats { diff --git a/crates/cellule-runtime/src/cell/actor/requests.rs b/crates/cellule-runtime/src/cell/actor/requests.rs index f41e6021..346b1ab3 100644 --- a/crates/cellule-runtime/src/cell/actor/requests.rs +++ b/crates/cellule-runtime/src/cell/actor/requests.rs @@ -467,6 +467,15 @@ pub(super) fn start_admitted_publication( return; }; let covered = coverage.len(); + // One physical native-group capture can represent several logical commands. + // The serialized publisher advances the exact contiguous Cell sequence. + let covered_commits = coverage.last().map_or(0, |queued| { + queued + .pending + .outcome() + .commit_sequence() + .saturating_sub(active.published_sequence) + }); let retained_bytes: u64 = coverage .iter() .map(|queued| queued.pending.retained_bytes()) @@ -503,6 +512,7 @@ pub(super) fn start_admitted_publication( let generation = active.generation; // Moving the publisher out of ActiveCell is the serialization token for // root preparation and CAS; no second object publisher can overtake it. + active.publishing_since = coverage.first().map(|queued| queued.submitted_at); let pool = pool.clone(); let published_next_due_ms = newest.pending.next_due_ms(); let published_commit_sequence = commit_sequence; @@ -633,6 +643,7 @@ pub(super) fn start_admitted_publication( total: newest_submitted_at.elapsed(), succeeded: result.is_ok(), commit_sequence, + covered_commits, }); tracing::debug!( target: "cellule_runtime::action", diff --git a/crates/cellule-runtime/src/cell/actor/runtime.rs b/crates/cellule-runtime/src/cell/actor/runtime.rs index 3e5ae0b4..045606c6 100644 --- a/crates/cellule-runtime/src/cell/actor/runtime.rs +++ b/crates/cellule-runtime/src/cell/actor/runtime.rs @@ -721,6 +721,18 @@ impl CellRuntime { self.inner.shutting_down.load(Ordering::Acquire) } + /// Samples committed command publication debt through the bounded actor lane. + /// This diagnostic performs no provider I/O and grants no writer authority. + pub async fn publication_progress(&self) -> crate::Result { + let (reply, response) = oneshot::channel(); + self.inner + .sender + .send(Message::PublicationProgress { reply }) + .await + .map_err(|_| Error::RuntimeClosed)?; + response.await.map_err(|_| Error::RuntimeClosed)? + } + /// Samples node-wide admission usage without waiting for actor work. #[must_use] pub fn stats(&self) -> CellRuntimeStats { diff --git a/crates/cellule-runtime/src/cell/actor/state.rs b/crates/cellule-runtime/src/cell/actor/state.rs index 7e417d57..7c117180 100644 --- a/crates/cellule-runtime/src/cell/actor/state.rs +++ b/crates/cellule-runtime/src/cell/actor/state.rs @@ -79,6 +79,9 @@ pub(super) struct BootstrapActivation { pub(super) type IdleTransferCandidates = Vec<(CellId, u64, i64, CatalogRole)>; pub(super) enum Message { + PublicationProgress { + reply: oneshot::Sender>, + }, Activate { cell: CellId, role: CatalogRole, @@ -279,6 +282,7 @@ pub(super) struct ActiveCell { pub(super) publisher: Option, pub(super) durability_submitter: CellDurabilitySubmitter, pub(super) publications: VecDeque, + pub(super) publishing_since: Option, pub(super) publication_bytes: u64, pub(super) unpublished_node_logs: usize, pub(super) queue: VecDeque, diff --git a/crates/cellule-runtime/src/cell/actor/task.rs b/crates/cellule-runtime/src/cell/actor/task.rs index 1a9848ca..2fbffc8e 100644 --- a/crates/cellule-runtime/src/cell/actor/task.rs +++ b/crates/cellule-runtime/src/cell/actor/task.rs @@ -355,7 +355,11 @@ pub(super) fn handle_message( movement_permits: &mut HashMap, next_generation: &mut u64, ) { - if !matches!(message, Message::Shutdown { .. }) && node_lease.check().is_err() { + if !matches!( + message, + Message::Shutdown { .. } | Message::PublicationProgress { .. } + ) && node_lease.check().is_err() + { for active in cells.values_mut() { active.coordination.step(CoordinationInput::Fence); fence_active(active); @@ -364,6 +368,29 @@ pub(super) fn handle_message( return; } match message { + Message::PublicationProgress { reply } => { + let now = std::time::Instant::now(); + let progress = CellPublicationProgress { + pending_publications: cells + .values() + .map(|active| active.coordination.publication_count()) + .sum(), + retained_capture_bytes: cells.values().map(|active| active.publication_bytes).sum(), + oldest_unpublished: cells + .values() + .flat_map(|active| { + active.publishing_since.into_iter().chain( + active + .publications + .front() + .map(|queued| queued.submitted_at), + ) + }) + .min() + .map(|oldest| now.saturating_duration_since(oldest)), + }; + let _ = reply.send(Ok(progress)); + } Message::Activate { cell, role, @@ -929,6 +956,9 @@ pub(super) fn handle_message( pub(super) fn reject_fenced_message(message: Message) { match message { + Message::PublicationProgress { reply } => { + let _ = reply.send(Err(Error::Fenced)); + } Message::Activate { reply, .. } => { let _ = reply.send(Err(Error::Fenced)); } diff --git a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs index 4711f20f..4fe65acb 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs @@ -129,6 +129,7 @@ pub(super) fn handle_activated( publisher: Some(*publisher), durability_submitter, publications: VecDeque::new(), + publishing_since: None, publication_bytes: 0, unpublished_node_logs: 0, queue: VecDeque::new(), diff --git a/crates/cellule-runtime/src/cell/actor/tasks/publication.rs b/crates/cellule-runtime/src/cell/actor/tasks/publication.rs index 87b41487..6665c27f 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/publication.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/publication.rs @@ -93,6 +93,11 @@ pub(super) fn handle_publication_admitted( total: elapsed, succeeded: false, commit_sequence: newest.pending.outcome().commit_sequence(), + covered_commits: newest + .pending + .outcome() + .commit_sequence() + .saturating_sub(active.published_sequence), }); } active.publisher = Some(*publisher); @@ -160,6 +165,7 @@ pub(super) fn handle_published( active.finish_task(effect_id, CoordinationEffect::Publication); active.last_work_at = std::time::Instant::now(); let object_published = result.is_ok(); + active.publishing_since = None; if object_published { // Control now names this commit, so the local mirror can answer a due // scan without reading the record back. diff --git a/crates/cellule-runtime/src/fleet/telemetry.rs b/crates/cellule-runtime/src/fleet/telemetry.rs index 9e2a0cb7..444792f5 100644 --- a/crates/cellule-runtime/src/fleet/telemetry.rs +++ b/crates/cellule-runtime/src/fleet/telemetry.rs @@ -32,6 +32,26 @@ pub struct PublicationTiming { pub succeeded: bool, /// Sequence used to correlate this observation with a request trace. pub commit_sequence: u64, + /// Logical commits included in this publication attempt. Count these only + /// when `succeeded` is true when calculating commands per selected root. + pub covered_commits: u64, +} + +/// Native follower timing for one dispatched append batch. +#[derive(Clone, Copy, Debug, Default)] +pub struct FollowerAppendTiming { + /// Time until the blocking worker starts; excludes its subsequent lane lock. + pub worker_queue: Duration, + /// Worker lifetime, including lane locking, verification, writes and barriers. + pub worker: Duration, + /// Time in append-file and rotation `sync_data` calls; excludes directory sync. + pub data_sync: Duration, + /// Number of attempted append-file and rotation barriers, including failures. + pub data_sync_calls: u64, + /// Number of input frames, including retries and already covered frames. + pub frames: u64, + /// Whether the append returned a durable receipt. + pub succeeded: bool, } /// Outcome of an actor-owned resident route lookup. @@ -206,6 +226,9 @@ pub trait CellTelemetry: Send + Sync { /// Records bytes sent to follower append lanes and whether every lane acknowledged them. fn node_log_append(&self, _acknowledged: bool, _bytes: u64) {} + /// Reports native follower work separately from peer HTTP and authorization. + fn follower_append(&self, _timing: FollowerAppendTiming) {} + /// Records a bounded-cardinality resident route result. fn resident_route(&self, _outcome: ResidentRouteOutcome) {} @@ -370,6 +393,12 @@ impl CellTelemetryHandle { } } + pub(crate) fn follower_append(&self, timing: FollowerAppendTiming) { + if let Some(telemetry) = self.inner.get() { + telemetry.follower_append(timing); + } + } + pub(crate) fn pressure_state(&self, state: PressureState) { if let Some(telemetry) = self.inner.get() { telemetry.pressure_state(state); diff --git a/crates/cellule-runtime/src/follower/mod.rs b/crates/cellule-runtime/src/follower/mod.rs index 10ec4d05..5cb9a195 100644 --- a/crates/cellule-runtime/src/follower/mod.rs +++ b/crates/cellule-runtime/src/follower/mod.rs @@ -8,8 +8,10 @@ use std::sync::{Arc, Mutex}; use bytes::Bytes; +use crate::fleet::telemetry::{CellTelemetryHandle, FollowerAppendTiming}; use crate::identity::SessionId; use crate::{Error, Result}; +use std::time::Instant; mod directory; mod inventory; @@ -151,6 +153,7 @@ pub struct FollowerStore { scan_counter: ScanCounter, admission: crate::fleet::admission::NodeAdmission, inventory_scope: [u8; 16], + telemetry: CellTelemetryHandle, } impl FollowerStore { @@ -183,6 +186,7 @@ impl FollowerStore { scan_counter: new_scan_counter(), admission: crate::fleet::admission::NodeAdmission::default(), inventory_scope: rand::random(), + telemetry: CellTelemetryHandle::default(), }) } @@ -198,6 +202,13 @@ impl FollowerStore { self } + /// Attaches a bounded telemetry sink before sharing the native store. + #[must_use] + pub fn with_telemetry(mut self, telemetry: CellTelemetryHandle) -> Self { + self.telemetry = telemetry; + self + } + #[cfg(test)] pub(crate) fn scan_count(&self) -> usize { self.scan_counter.load(Ordering::Relaxed) @@ -263,7 +274,19 @@ impl FollowerStore { .ok_or(Error::Node("follower append byte count overflow"))?; let admission = self.admission.clone(); let lanes = Arc::clone(&self.lanes); + let queued_at = Instant::now(); + let telemetry = self.telemetry.clone(); + let frame_count = frames.len() as u64; tokio::task::spawn_blocking(move || { + let mut observation = AppendObservation { + started: Instant::now(), + telemetry, + timing: FollowerAppendTiming { + worker_queue: queued_at.elapsed(), + frames: frame_count, + ..FollowerAppendTiming::default() + }, + }; let retained = retained .lock() .map_err(|_| Error::Node("follower disk reservation lock poisoned"))?; @@ -284,6 +307,7 @@ impl FollowerStore { &index_used, &mut state, &scan_counter, + &mut observation.timing, ) })(); let resize = @@ -307,6 +331,7 @@ impl FollowerStore { } } } + observation.timing.succeeded = result.is_ok(); result }) .await @@ -576,5 +601,18 @@ impl FollowerStore { } } +struct AppendObservation { + started: Instant, + telemetry: CellTelemetryHandle, + timing: FollowerAppendTiming, +} + +impl Drop for AppendObservation { + fn drop(&mut self) { + self.timing.worker = self.started.elapsed(); + self.telemetry.follower_append(self.timing); + } +} + #[cfg(test)] mod tests; diff --git a/crates/cellule-runtime/src/follower/records/append.rs b/crates/cellule-runtime/src/follower/records/append.rs index 1ac7bb9c..19d60ec0 100644 --- a/crates/cellule-runtime/src/follower/records/append.rs +++ b/crates/cellule-runtime/src/follower/records/append.rs @@ -3,6 +3,10 @@ use super::*; use std::collections::BTreeSet; +#[expect( + clippy::too_many_arguments, + reason = "the append binding and its caller-owned timing remain explicit" +)] pub(in crate::follower) fn append_sync( root: &Path, lane: Lane, @@ -12,6 +16,7 @@ pub(in crate::follower) fn append_sync( index_used: &Arc>, state: &mut Option, scan_counter: &ScanCounter, + timing: &mut FollowerAppendTiming, ) -> Result { validate_lane(lane)?; let directory = lane_directory(root, lane); @@ -104,7 +109,7 @@ pub(in crate::follower) fn append_sync( if file.metadata()?.len() > 0 && file.metadata()?.len().saturating_add(record_bytes) > ROTATE_BYTES { - file.sync_data()?; + sync_append(&file, timing)?; drop(file); let destination = rotate_open(&chunks, &open_path, open_first, open_last)?; relocate_records( @@ -141,7 +146,7 @@ pub(in crate::follower) fn append_sync( }); } if !pending.is_empty() { - file.sync_data()?; + sync_append(&file, timing)?; state .records .extend(pending.into_iter().map(|record| (record.sequence, record))); @@ -164,6 +169,14 @@ pub(in crate::follower) fn append_sync( durable_through, }) } + +fn sync_append(file: &std::fs::File, timing: &mut FollowerAppendTiming) -> Result<()> { + let started = Instant::now(); + let result = file.sync_data(); + timing.data_sync += started.elapsed(); + timing.data_sync_calls = timing.data_sync_calls.saturating_add(1); + result.map_err(Into::into) +} pub(in crate::follower) fn seal_sync( root: &Path, lane: Lane, diff --git a/crates/cellule-runtime/src/follower/tests/append.rs b/crates/cellule-runtime/src/follower/tests/append.rs index ba7b533e..1a15ff65 100644 --- a/crates/cellule-runtime/src/follower/tests/append.rs +++ b/crates/cellule-runtime/src/follower/tests/append.rs @@ -2,6 +2,66 @@ use super::*; +#[tokio::test] +async fn native_append_timing_distinguishes_durable_batches_from_retries() { + #[derive(Default)] + struct Timings(Mutex>); + impl crate::fleet::telemetry::CellTelemetry for Timings { + fn follower_append(&self, timing: crate::fleet::telemetry::FollowerAppendTiming) { + self.0.lock().unwrap().push(timing); + } + } + let sink = Arc::new(Timings::default()); + let limits = cellule_ltx::Limits::default(); + let source = tempfile::TempDir::new().unwrap(); + let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); + database + .transaction(|tx| { + tx.execute_batch("CREATE TABLE values_(v); INSERT INTO values_ VALUES(1)") + }) + .unwrap(); + let capture = database.capture().unwrap(); + let segment = capture.segments.first().unwrap(); + let root = tempfile::TempDir::new().unwrap(); + let store = FollowerStore::open( + root.path().to_owned(), + limits, + cellule_ltx::DiskBudget::new(1 << 30), + ) + .unwrap() + .with_telemetry(crate::fleet::telemetry::CellTelemetryHandle::from_sink( + sink.clone(), + )); + let leader = SessionId::from_bytes([1; 16]); + let frames = vec![frame(1, segment, limits), frame(2, segment, limits)]; + assert_eq!( + store + .append(leader, 2, frames.clone(), 0) + .await + .unwrap() + .durable_through, + 2 + ); + assert_eq!( + store + .append(leader, 2, frames, 0) + .await + .unwrap() + .durable_through, + 2 + ); + let observed = sink.0.lock().unwrap(); + assert_eq!(observed.len(), 2); + assert!( + observed + .iter() + .all(|timing| timing.succeeded && timing.frames == 2) + ); + assert_eq!(observed[0].data_sync_calls, 1); + assert_eq!(observed[1].data_sync_calls, 0); + assert!(observed[0].data_sync <= observed[0].worker); +} + #[tokio::test] async fn cordon_blocks_new_lanes_while_existing_tail_appends_and_restart_remain_safe() { let limits = cellule_ltx::Limits::default(); diff --git a/crates/cellule-runtime/src/lib.rs b/crates/cellule-runtime/src/lib.rs index f536e53b..a4afe139 100644 --- a/crates/cellule-runtime/src/lib.rs +++ b/crates/cellule-runtime/src/lib.rs @@ -46,7 +46,7 @@ pub mod read_policy; pub mod recovery; pub mod registry; -pub use cell::actor::{CellRuntime, CellRuntimeStats}; +pub use cell::actor::{CellPublicationProgress, CellRuntime, CellRuntimeStats}; pub use cell::catalog::CatalogRole; pub use cell::executor::{MutationIdentity, Resolution}; diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index 3734ba45..3eaebe29 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -170,6 +170,11 @@ pub struct NodeDurability { } impl NodeDurability { + /// Observes the epoch's durability frontiers; this does not issue a proof. + pub fn progress(&self) -> Result { + self.gate.progress() + } + /// Creates one node-log durability epoch over its gate, shipper, authority, /// transport, and node lease. #[must_use] diff --git a/crates/cellule-runtime/src/node/log/mod.rs b/crates/cellule-runtime/src/node/log/mod.rs index 6b14d0b8..37dcd941 100644 --- a/crates/cellule-runtime/src/node/log/mod.rs +++ b/crates/cellule-runtime/src/node/log/mod.rs @@ -141,6 +141,29 @@ pub struct DurabilityGate { changed: Arc, } +/// One consistent observation of an epoch's durability frontiers. +/// +/// Observations grant no durability, recovery, rotation, or deletion authority. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct NodeLogProgress { + /// Epoch these frontiers describe. + pub log_epoch: u64, + /// Highest issued node-log sequence. + pub issued_through: u64, + /// Highest sequence fsynced by every enrolled follower. + pub follower_proven_through: u64, + /// Contiguous prefix covered by authoritative object publication. + pub tiered_through: u64, + /// Issued sequences beyond the contiguous object-covered prefix. + pub pending_object_sequences: u64, + /// Whether authoritative activation enabled fleet proofs. + pub fleet_active: bool, + /// Whether issuance has stopped for rotation. + pub rotating: bool, + /// Whether the gate has been fenced. + pub fenced: bool, +} + struct GateState { leader_session: SessionId, leader_node: NodeId, @@ -158,6 +181,27 @@ struct GateState { } impl DurabilityGate { + /// Samples all frontiers under one lock without issuing a proof. + pub fn progress(&self) -> Result { + let state = self.lock()?; + let issued_through = state.next_sequence.saturating_sub(1); + Ok(NodeLogProgress { + log_epoch: state.log_epoch, + issued_through, + follower_proven_through: state + .members + .iter() + .map(|member| state.follower_through.get(member).copied().unwrap_or(0)) + .min() + .unwrap_or(0), + tiered_through: state.tiered_through, + pending_object_sequences: issued_through.saturating_sub(state.tiered_through), + fleet_active: state.fleet_active, + rotating: state.rotating, + fenced: state.fenced, + }) + } + /// Creates one inactive gate for the exact recruited follower ensemble. pub fn new( leader_session: SessionId, diff --git a/crates/cellule-runtime/src/node/log/tests.rs b/crates/cellule-runtime/src/node/log/tests.rs index 6cd7618f..c88fe52c 100644 --- a/crates/cellule-runtime/src/node/log/tests.rs +++ b/crates/cellule-runtime/src/node/log/tests.rs @@ -134,6 +134,28 @@ fn node(byte: u8) -> NodeId { NodeId::from_bytes([byte; 16]) } +#[test] +fn progress_keeps_follower_proof_distinct_from_contiguous_object_coverage() { + let gate = DurabilityGate::new(session(1), node(1), 2, [node(3), node(4)]).unwrap(); + let first = gate.issue(2).unwrap(); + let second = gate.issue(1).unwrap(); + gate.acknowledge(node(3), 3).unwrap(); + gate.acknowledge(node(4), 2).unwrap(); + gate.prove_object(second).unwrap(); + let progress = gate.progress().unwrap(); + assert_eq!(progress.issued_through, 3); + assert_eq!(progress.follower_proven_through, 2); + assert_eq!(progress.tiered_through, 0); + assert_eq!(progress.pending_object_sequences, 3); + assert!(!progress.fleet_active); + assert!(gate.proof(first).unwrap().is_none()); + gate.prove_object(first).unwrap(); + assert_eq!(gate.progress().unwrap().pending_object_sequences, 0); + assert_eq!(gate.progress().unwrap().tiered_through, 3); + gate.fence(); + assert!(gate.progress().unwrap().fenced); +} + #[tokio::test] async fn fleet_requires_activation_and_every_follower() { let gate = DurabilityGate::new(session(1), node(1), 2, [node(3), node(4)]).unwrap(); diff --git a/crates/cellule-runtime/src/publication/tests.rs b/crates/cellule-runtime/src/publication/tests.rs index 06cb9a60..2e18662d 100644 --- a/crates/cellule-runtime/src/publication/tests.rs +++ b/crates/cellule-runtime/src/publication/tests.rs @@ -384,8 +384,8 @@ async fn pressure_append_avoids_intermediate_root_metadata() { database.close().unwrap(); assert_eq!( requests, - (7, 1), - "only final directory, root and lineage metadata is retained" + (4, 1), + "two packed segments, the final root and lineage are retained" ); } } diff --git a/crates/cellule-runtime/src/recovery/retention/mod.rs b/crates/cellule-runtime/src/recovery/retention/mod.rs index 402e2fe5..2c098d50 100644 --- a/crates/cellule-runtime/src/recovery/retention/mod.rs +++ b/crates/cellule-runtime/src/recovery/retention/mod.rs @@ -545,7 +545,10 @@ fn immutable_candidate(application_prefix: &Path, location: &Path) -> bool { && lower_hex(incarnation, 32) && object.rsplit_once('.').is_some_and(|(digest, extension)| { lower_hex(digest, 64) - && matches!(extension, "ltx" | "index" | "dir" | "root" | "bundle") + && matches!( + extension, + "ltx" | "index" | "dir" | "root" | "bundle" | "pack" + ) }) } _ => false, diff --git a/crates/cellule-runtime/tests/runtime/backup.rs b/crates/cellule-runtime/tests/runtime/backup.rs index e98fe286..49288edb 100644 --- a/crates/cellule-runtime/tests/runtime/backup.rs +++ b/crates/cellule-runtime/tests/runtime/backup.rs @@ -199,7 +199,7 @@ async fn pin_verifies_roots_and_fails_closed_when_a_dependency_is_missing() { .await .unwrap() .into_iter() - .find(|object| object.kind == CellObjectKind::Ltx) + .find(|object| matches!(object.kind, CellObjectKind::Ltx | CellObjectKind::Packed)) .unwrap(); let missing_path = layout.incarnation_object_path( cell.as_bytes(), diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission_batching.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission_batching.rs index 1a2ac176..b754673b 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission_batching.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/admission_batching.rs @@ -95,6 +95,10 @@ async fn publication_admission_batches_commits_that_arrive_while_waiting() { } responses.wait_for_responses(4).await; + let progress = runtime.publication_progress().await.unwrap(); + assert_eq!(progress.pending_publications, 4); + assert!(progress.retained_capture_bytes > 0); + assert!(progress.oldest_unpublished.is_some()); assert!( responses .0 @@ -108,6 +112,10 @@ async fn publication_admission_batches_commits_that_arrive_while_waiting() { .await .unwrap() .unwrap(); + let progress = runtime.publication_progress().await.unwrap(); + assert_eq!(progress.pending_publications, 0); + assert_eq!(progress.retained_capture_bytes, 0); + assert_eq!(progress.oldest_unpublished, None); runtime.shutdown().await.unwrap(); let released = CellAuthority::new(fixture.layout.clone()) diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/group.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/group.rs index 2fba7da3..6c8a6a9a 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability/group.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/group.rs @@ -194,6 +194,17 @@ async fn queued_fleet_groups_recover_and_backlog_refuses_before_sql() { assert_eq!(frames[0].scope().commit_sequence, 4); assert_eq!(frames[1].first_commit_sequence(), 5); assert_eq!(frames[1].scope().commit_sequence, 6); + assert_eq!( + responses + .1 + .lock() + .unwrap() + .iter() + .filter(|timing| timing.succeeded) + .map(|timing| timing.covered_commits) + .sum::(), + 6 + ); assert_eq!( &responses.0.lock().unwrap()[..6], &[CommandResponseSource::Fleet; 6] @@ -431,6 +442,13 @@ async fn queued_native_mutations_share_one_capture_and_exact_root_proof() { .collect::>(), vec![1, 5] ); + assert_eq!( + publications + .iter() + .map(|timing| timing.covered_commits) + .collect::>(), + vec![1, 4] + ); assert_eq!( root.position.txid, 3, "one SQLite commit/capture for the queued group" diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/ownership/prefix.rs b/crates/cellule-runtime/tests/runtime/lifecycle/ownership/prefix.rs index 495791fa..b47a0510 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/ownership/prefix.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/ownership/prefix.rs @@ -162,7 +162,7 @@ async fn cached_root_metadata_cannot_hide_a_missing_or_corrupt_current_dependenc runtime.shutdown().await.unwrap(); let object = objects .iter() - .find(|object| object.kind == CellObjectKind::Ltx) + .find(|object| matches!(object.kind, CellObjectKind::Ltx | CellObjectKind::Packed)) .unwrap(); let path = fixture.layout.incarnation_object_path( &root.cell, diff --git a/crates/cellule-store/src/observation/mod.rs b/crates/cellule-store/src/observation/mod.rs index b85f301a..98225d51 100644 --- a/crates/cellule-store/src/observation/mod.rs +++ b/crates/cellule-store/src/observation/mod.rs @@ -161,6 +161,15 @@ pub struct StorageObservation { /// Receives bounded backend lifecycle events. pub trait StorageObserver: Send + Sync { + /// Selects a bounded application-owned observer for this object or prefix. + /// + /// The returned observer receives the complete operation, including streamed + /// reads, cancellation, and multipart parts. Selection occurs once; do not + /// retain object identities as metric labels. `None` uses this observer. + fn for_object(&self, _location: Option<&Path>) -> Option> { + None + } + /// Marks a newly active logical backend operation. fn started(&self, operation: StorageOperation); @@ -200,7 +209,8 @@ impl ObjectStore for ObservedObjectStore { options: PutOptions, ) -> object_store::Result { let bytes = saturating_u64(payload.content_length()); - let mut observation = ActiveObservation::new(StorageOperation::Put, &self.observer); + let mut observation = + ActiveObservation::for_object(StorageOperation::Put, &self.observer, Some(location)); let result = self.inner.put_opts(location, payload, options).await; observation.finish_result(&result, 0, bytes); result @@ -211,15 +221,18 @@ impl ObjectStore for ObservedObjectStore { location: &Path, options: PutMultipartOptions, ) -> object_store::Result> { - let mut observation = - ActiveObservation::new(StorageOperation::MultipartStart, &self.observer); + let mut observation = ActiveObservation::for_object( + StorageOperation::MultipartStart, + &self.observer, + Some(location), + ); let result = self.inner.put_multipart_opts(location, options).await; match result { Ok(upload) => { observation.finish(StorageOutcome::Success, 0, 0); Ok(Box::new(ObservedMultipartUpload { inner: upload, - observer: Arc::clone(&self.observer), + observer: Arc::clone(&observation.observer), })) } Err(error) => { @@ -241,7 +254,8 @@ impl ObjectStore for ObservedObjectStore { } else { StorageOperation::Get }; - let mut observation = ActiveObservation::new(operation, &self.observer); + let mut observation = + ActiveObservation::for_object(operation, &self.observer, Some(location)); let result = self.inner.get_opts(location, options).await; let result = match result { Ok(result) => result, @@ -275,7 +289,8 @@ impl ObjectStore for ObservedObjectStore { location: &Path, ranges: &[Range], ) -> object_store::Result> { - let mut observation = ActiveObservation::new(StorageOperation::Range, &self.observer); + let mut observation = + ActiveObservation::for_object(StorageOperation::Range, &self.observer, Some(location)); let result = self.inner.get_ranges(location, ranges).await; let bytes_read = result.as_ref().map_or(0, |bodies| { bodies.iter().fold(0_u64, |total, body| { @@ -290,12 +305,14 @@ impl ObjectStore for ObservedObjectStore { &self, locations: BoxStream<'static, object_store::Result>, ) -> BoxStream<'static, object_store::Result> { - let observation = ActiveObservation::new(StorageOperation::Delete, &self.observer); + let observation = + ActiveObservation::for_object(StorageOperation::Delete, &self.observer, None); observe_stream(self.inner.delete_stream(locations), observation) } fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { - let observation = ActiveObservation::new(StorageOperation::List, &self.observer); + let observation = + ActiveObservation::for_object(StorageOperation::List, &self.observer, prefix); observe_stream(self.inner.list(prefix), observation) } @@ -304,12 +321,14 @@ impl ObjectStore for ObservedObjectStore { prefix: Option<&Path>, offset: &Path, ) -> BoxStream<'static, object_store::Result> { - let observation = ActiveObservation::new(StorageOperation::List, &self.observer); + let observation = + ActiveObservation::for_object(StorageOperation::List, &self.observer, prefix); observe_stream(self.inner.list_with_offset(prefix, offset), observation) } async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { - let mut observation = ActiveObservation::new(StorageOperation::List, &self.observer); + let mut observation = + ActiveObservation::for_object(StorageOperation::List, &self.observer, prefix); let result = self.inner.list_with_delimiter(prefix).await; observation.finish_result(&result, 0, 0); result @@ -321,7 +340,8 @@ impl ObjectStore for ObservedObjectStore { to: &Path, options: CopyOptions, ) -> object_store::Result<()> { - let mut observation = ActiveObservation::new(StorageOperation::Copy, &self.observer); + let mut observation = + ActiveObservation::for_object(StorageOperation::Copy, &self.observer, Some(to)); let result = self.inner.copy_opts(from, to, options).await; observation.finish_result(&result, 0, 0); result @@ -342,8 +362,11 @@ impl ObservedMultipartStore { #[async_trait::async_trait] impl MultipartStore for ObservedMultipartStore { async fn create_multipart(&self, path: &Path) -> object_store::Result { - let mut observation = - ActiveObservation::new(StorageOperation::MultipartStart, &self.observer); + let mut observation = ActiveObservation::for_object( + StorageOperation::MultipartStart, + &self.observer, + Some(path), + ); let result = self.inner.create_multipart(path).await; observation.finish_result(&result, 0, 0); result @@ -354,8 +377,11 @@ impl MultipartStore for ObservedMultipartStore { path: &Path, options: PutMultipartOptions, ) -> object_store::Result { - let mut observation = - ActiveObservation::new(StorageOperation::MultipartStart, &self.observer); + let mut observation = ActiveObservation::for_object( + StorageOperation::MultipartStart, + &self.observer, + Some(path), + ); let result = self.inner.create_multipart_opts(path, options).await; observation.finish_result(&result, 0, 0); result @@ -369,8 +395,11 @@ impl MultipartStore for ObservedMultipartStore { data: PutPayload, ) -> object_store::Result { let bytes = saturating_u64(data.content_length()); - let mut observation = - ActiveObservation::new(StorageOperation::MultipartPart, &self.observer); + let mut observation = ActiveObservation::for_object( + StorageOperation::MultipartPart, + &self.observer, + Some(path), + ); let result = self.inner.put_part(path, id, part_idx, data).await; observation.finish_result(&result, 0, bytes); result @@ -382,16 +411,22 @@ impl MultipartStore for ObservedMultipartStore { id: &MultipartId, parts: Vec, ) -> object_store::Result { - let mut observation = - ActiveObservation::new(StorageOperation::MultipartComplete, &self.observer); + let mut observation = ActiveObservation::for_object( + StorageOperation::MultipartComplete, + &self.observer, + Some(path), + ); let result = self.inner.complete_multipart(path, id, parts).await; observation.finish_result(&result, 0, 0); result } async fn abort_multipart(&self, path: &Path, id: &MultipartId) -> object_store::Result<()> { - let mut observation = - ActiveObservation::new(StorageOperation::MultipartAbort, &self.observer); + let mut observation = ActiveObservation::for_object( + StorageOperation::MultipartAbort, + &self.observer, + Some(path), + ); let result = self.inner.abort_multipart(path, id).await; observation.finish_result(&result, 0, 0); result @@ -451,6 +486,15 @@ struct ActiveObservation { } impl ActiveObservation { + fn for_object( + operation: StorageOperation, + observer: &Arc, + location: Option<&Path>, + ) -> Self { + let selected = observer.for_object(location); + Self::new(operation, selected.as_ref().unwrap_or(observer)) + } + fn new(operation: StorageOperation, observer: &Arc) -> Self { observer.started(operation); Self { diff --git a/crates/cellule-store/src/observation/tests.rs b/crates/cellule-store/src/observation/tests.rs index 972107e1..13b2f10a 100644 --- a/crates/cellule-store/src/observation/tests.rs +++ b/crates/cellule-store/src/observation/tests.rs @@ -43,6 +43,56 @@ fn observed_store(observer: &Arc) -> Store { .with_storage_observer(Arc::clone(observer) as Arc) } +struct RoutedObserver(Arc); + +impl StorageObserver for RoutedObserver { + fn for_object(&self, location: Option<&Path>) -> Option> { + assert!(location.is_some()); + Some(self.0.clone()) + } + + fn started(&self, _: StorageOperation) { + panic!("operation was not routed"); + } + + fn finished(&self, _: StorageObservation) { + panic!("completion was not routed"); + } +} + +#[tokio::test] +async fn selected_observer_owns_stream_cancellation_and_multipart_lifetime() { + let target = Arc::new(RecordingObserver::default()); + let store = Store::new(Arc::new(InMemory::new())) + .with_storage_observer(Arc::new(RoutedObserver(target.clone()))); + let path = Path::from("bounded-family/object"); + let mut upload = store.inner().put_multipart(&path).await.unwrap(); + upload + .put_part(Bytes::from_static(b"payload").into()) + .await + .unwrap(); + upload.complete().await.unwrap(); + let result = store.inner().get(&path).await.unwrap(); + assert_eq!(target.active(StorageOperation::Get), 1); + drop(result); + assert_eq!(target.active(StorageOperation::Get), 0); + let observations = target.observations(); + assert!(observations.iter().any(|o| o.operation == StorageOperation::Get && o.outcome == StorageOutcome::Cancelled)); + for operation in [ + StorageOperation::MultipartStart, + StorageOperation::MultipartPart, + StorageOperation::MultipartComplete, + ] { + assert_eq!( + observations + .iter() + .filter(|o| o.operation == operation) + .count(), + 1 + ); + } +} + struct TestMultipartStore; #[async_trait::async_trait] diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md new file mode 100644 index 00000000..26b65905 --- /dev/null +++ b/docs/write-performance-delivery.md @@ -0,0 +1,115 @@ +# Write performance implementation and verification + +The implementation packs small publication dependencies and provides a pinned +Docker comparison with exact retry and cold-state audits. **Celld write parity +has not been established.** The [proposal](write-performance-proposal.md) remains +the acceptance contract; completing tests or a load run does not pass its gates. + +## Delivered behavior + +| Change | Measurable result | Preserved contract | +| --- | --- | --- | +| Small native LTX and index share one `.pack` | One dependency PUT instead of two; at most 256 KiB | Exact native bytes, body/index digests, complete object digest | +| Small directory leaf lives in the root | Removes its separate PUT, GET and cached-origin HEAD | Canonical leaf validation; 2 KiB leaf and 32 KiB root bounds | +| Packed compaction input supplies both scratch streams | One full GET per selected pack instead of full plus range GET | Complete verification; bounded transfer and file ownership through cancellation | +| Window telemetry separates response, proof and publication | Logical commands per selected root; capture/checkpoint, worker, peer and sync histograms | Counters and frontiers confer no authority or proof | +| Storage families distinguish owner/receiver enrollment | Summed enrollment GET cost across all three nodes | Each signed peer message still uses fresh authorization | +| Docker runner and reports preserve failures | Source/binary identities, fresh provider volumes, scheduled-arrival latency, all-ACK audits | Errors, drops, unissued offers, provider failures and failed drain cannot pass | + +An ordinary small root needs two immutable PUTs plus lineage and fenced Cell +selection: **four successful PUTs instead of six**. A small scheduled +compaction composed with an append still needs two packs, the final root, +lineage and selection: **five PUTs**. Node coverage and maintenance remain in +the window numerator. M1's universal four-PUT gate has therefore not passed. + +The current root development format is version 2. The +[format specification](../crates/cellule-ltx/docs/packed-root-format.md) covers +all readers, producers, sparse-read locators, recovery inventories, backup and +collection paths. There is no legacy decoding or automatic migration. + +## Milestone status + +| Milestone | Implementation | Exit gate | +| --- | --- | --- | +| M0 | Measurement and comparison harness delivered | Three A/A capacity pairs unverified; storage API totals reconcile, but SDK-internal HTTP retries need provider telemetry | +| M1 | Packs, inline leaves and bounded compaction spooling delivered | Ordinary append meets four PUTs; composed compaction needs five. Paired cold/sparse-read guardrail unverified | +| M2 | Existing native grouping and per-Cell coalescing preserved | Shared node publication coordinator not implemented; 0.25 publication PUTs/command not achieved | +| M3 | Fresh enrollment roles, peer phases and follower append measured | Signed append grants and durable grant fences not implemented; 0.05 enrollment GETs/command not achieved | +| M4 | Existing authority-pinned Cell roots remain the object proof | Bundle coverage proof, transfer and collection protocol not implemented | +| M5 | All-ACK warm/cold audits and five-minute main characterization completed | Paired repetitions, read/failure/overload matrix and absolute/relative parity unverified | + +## Evidence + +The fixture is **SQL application parity**, not the user's bounded KV workload: +1,000 Cells, 96-byte values, INSERT plus in-command SELECT, and a two-hour +durable request/result ledger in both applications. Nodes have 8-CPU/16-GiB +ceilings and tmpfs state, RustFS has 2 CPUs/2 GiB, and the client has 4 CPUs/4 GiB +on one 8-CPU/16-GiB Linux VM. Summed CPU ceilings exceed VM capacity. This +profile qualifies neither device persistence nor independent-node isolation. + +Latest main is `18eff0f7af47fac09b993157bb444e582072d7cf`, which merged PR 65. +Its production source matches the audited `397f500a` foundation; three additional +main files are design documents. The baseline explicitly overlays measurement +hooks. Celld is v0.6.1, `f2bf648663a610eefde71f3547ad61e9b896b1f0`, using the +pinned container digest. Both framework arms use byte-identical clients/auditors. + +Latest-main Fleet offered 100 writes/s for 300 seconds after 30 seconds warmup. +It completed **99.993 writes/s**, with **868.1 ms scheduled p99**, zero errors +and zero drops. Every one of its **34,001 acknowledged commands** passed GET +and exact-command retry audits while warm and after all original node containers +were removed. The original fleet drained in **7.737 seconds**. This point fails +the 50-ms Fleet latency gate and publication stability. + +Its window measured **6.688 successful PUTs/command**, **3.327 GET attempts/command +including ranges**, and **1.256 fresh enrollment GETs/command**, summed over +owner and followers. Unpublished bytes and oldest publication age had positive +slopes over the final three minute segments. The provider was at its two-CPU +ceiling during much of the run. Fleet proof wait averaged 88.0 ms, response +confirmation 0.290 ms, capture 1.047 ms, and tmpfs follower data sync about +0.0007 ms. This points to publication and provider/peer pressure in this profile; +it does not establish the bottleneck for NVMe or managed object storage. + +Candidate and matched celld results will be recorded after their complete audits. +No candidate speedup is inferred from the main result. + +## Reproduce and inspect + +Follow the [harness instructions](../scripts/perf/README.md). Build each revision +into a fresh external directory. Run serially and preserve failed cases. +Inspect `case.json`, `build.json`, `storage-format-smoke.json`, `summary.json` +and generated `report.json`. Content-based cache namespaces and persisted-root +checks prevent stale codec reuse. Source contains reusable drivers and compact +conclusions; caches, volumes, journals, metric windows, binaries and logs stay +outside Git. Each retained provider volume has `store-data/volume.json`. + +`scripts/perf/compare.py` accepts an external JSON matrix containing `baseline`, +`candidate` and `celld` lists of case directories. It rejects different workload, +resources, images or client identities. Matching completion rates cannot establish +the full proposal from one pair. Reports include full and range GETs, copy and +multipart calls. SDK-internal retries need provider telemetry for exact HTTP counts. + +## Cutover and rollback + +1. Use fresh isolated prefixes for qualification and recreatable development data. +2. For a persisted format cutover, stop admission, drain accepted commands and + preserve the bucket and recovery logs. Upgrade every root reader, producer, + recovery, backup and collection worker together. +3. Retained version-1 data needs a separately verified logical export/rebuild + using the old binary. No automatic conversion route is delivered here. +4. A rollback binary cannot read version-2 roots. Use a verified logical + export/rebuild if available; otherwise preserve artifacts and roll forward. + Bucket listing and completed uploads never select authority. + +## Remaining delivery + +Complete M0's reproducibility gate before attributing sustainable-rate changes +to the framework. Then implement M2's bounded node publication coordinator, +complete cross-Cell inventory and per-Cell reconciliation. Measure the selection +floor before deciding M4. M3 needs scoped signed grants and receiver fences with +seal, retire, restart, expiry and lease-loss tests; a TTL cache of consumed peer +verifiers does not implement that contract. + +Finally run three paired five-minute repetitions, bounded KV, read-only and +mixed/hot-read guardrails, 1.5× overload with immediate recovery, owner loss +before materialization, and device durability. Preserve failures and report +the measured gap until every required gate passes. diff --git a/docs/write-performance-proposal.md b/docs/write-performance-proposal.md index c2ab8925..6c33fcb9 100644 --- a/docs/write-performance-proposal.md +++ b/docs/write-performance-proposal.md @@ -1,6 +1,9 @@ # Cellule write performance proposal and delivery plan Status: proposed implementation plan, 2026-10-06. Write parity is not achieved. +Implementation and actual verification are tracked in the +[delivery report](write-performance-delivery.md). Its partial milestone status +does not relax the acceptance gates below. The implementation baseline is PR [65](https://github.com/crabbuild/cellule/pull/65) at `397f500a`, against main `0813f974` and celld v0.6.1 `f2bf6486`. The [performance plan](performance-plan.md) retains the earlier audit and local diff --git a/scripts/perf/README.md b/scripts/perf/README.md new file mode 100644 index 00000000..e1d80508 --- /dev/null +++ b/scripts/perf/README.md @@ -0,0 +1,113 @@ +# Docker write verification + +This harness runs the same SQL application on Cellule and pinned celld v0.6.1. +It uses 1,000 Cells, 96-byte values, INSERT plus SELECT in one transaction, +and a two-hour durable request/result ledger. It is **SQL application parity**; +it does not reproduce a bounded KV upsert benchmark. + +Use a dedicated Linux Docker context. The current shared-VM profile gives +nodes an 8-CPU/16-GiB ceiling, tmpfs state of 4 GiB, a 2-CPU/2-GiB RustFS +provider, and a 4-CPU/4-GiB client. These ceilings exceed the shared VM's total +CPU; report contention. tmpfs does not qualify physical-device durability. +HTTP endpoints, certificates, credentials, and placement are fixture policy. +The credentials in these scripts are synthetic and used only by this fixture. + +## Build and run + +All artifacts, build caches, binaries, logs and audit journals go +outside the checkout. Each case gets a fresh labeled Docker volume for RustFS +data on the Linux filesystem, retained for investigation. Its name is recorded +in `store-data/volume.json`; it is never a host filesystem bind mount. The build records every exported source digest, the +adapted fixture source, immutable binary hashes and pinned image digests. +Never reuse an artifact directory for a new build. +Build caches are partitioned by every adapted source digest and the compiler +image. Before measurement, the runner reads a persisted root from the provider +and checks its codec version against the exported source; the small-root +candidate must also contain a packed dependency. This catches stale local Cargo +libraries that binary/source manifests alone would miss. + +```sh +python3 scripts/perf/build.py --context dedicated-linux \ + --artifacts /absolute/external/candidate +python3 scripts/perf/run.py cellule-fleet celld-fleet \ + --context dedicated-linux --artifacts /absolute/external/candidate \ + --tag paired --seconds 300 --warmup 30 --repetitions 3 \ + --write-rates 100 200 500 15000 +python3 scripts/perf/report.py /absolute/external/candidate/cellule-fleet-parity-paired-r1 +``` + +Put case names before `--write-rates`, whose list extends until the next option. +Run sustainable-rate searches before target stress. A completed run is not a +qualified run. Failed runs are retained and the runner exits nonzero if any +case, audit or drain is incomplete. Overloaded completion rates are not capacity. + +For latest main after PR 65, retain the same measurement hooks +and workload while leaving publication and storage codecs at the baseline: + +```sh +python3 scripts/perf/build.py --context dedicated-linux \ + --artifacts /absolute/external/baseline \ + --framework-ref 18eff0f7af47fac09b993157bb444e582072d7cf \ + --measurement-overlay +``` + +The overlay is explicitly recorded and supported for that revision and the +audited PR 65 foundation `397f500a39d0cd3a84724a82fe66d6a56e76f726`. +Their production source is identical; main added three design documents. +The overlay build is not an unmodified main build. Client and auditor hashes must match across +comparisons. Serialize runs; a context lock prevents concurrent harness runs. +The harness owns only its named, labeled provider and client containers. +Its loopback ports must be free of other workloads. + +For a read-only point, supply `--write-rates 0 --read-rate N`. For a mixed +point, use the same offered write rate on both systems and add `--read-rate N`. +Add `--hot-read-cells 10` for the 1% hot-read case. Record qualification of +read-only and mixed profiles separately; an aggregate read/write rate is not a +read-capacity result. + +After establishing qualified capacity, add `--overload-capacity N`. The runner +then offers `1.5 × N` for 60 seconds and immediately `0.5 × N` for 30 seconds, +with no intervening warmup. Both cohorts enter the warm and cold audits. +The report separates these phases from steady capacity and records the total +fleet drain duration. The supplied reference rate still requires paired +qualification, and HTTP evidence alone does not prove every refusal preceded SQL. + +## Evidence contract + +Latency begins at scheduled arrival. Journals include every attempted request; +counts include unissued offers, client queue drops, HTTP errors and late +completions. The client samples raw cumulative histograms and provider counters +at window start, every minute, and window end; the report subtracts counters +rather than percentiles. Endpoint session, schema and counter resets invalidate +the metric window. All storage families sum to provider operation counters. +Reports include range GETs, copy and multipart operations separately; a full-GET +count alone hides compaction traffic. These observations count storage API calls. +Retries inside a provider SDK require provider-side telemetry for exact HTTP +attempt counts. Families distinguish immutable data, Cell authority, node authority, owner and +receiver enrollment, and other work. Enrollment scopes follow each GET through +stream completion. Compaction and conditional PUT roles need finer attribution; +all attempts and successful PUTs remain in the reported provider totals. +Schema 3 exports nonzero cumulative histogram buckets with their original +indices and fixed bound, preserving 100-us resolution and overflow. The client +primes exporters and sampling connections before warmup. Use `--telemetry off` +only to measure exporter overhead; that diagnostic cannot qualify. Compare +response, confirmation, fleet-proof, capture/checkpoint, and publication timings +separately instead of attributing publication time to the command ACK. + +Warm and bucket-only cold audits GET every acknowledged mutation and retry +**every original command**, checking its exact stored output and sequence. +Original owner and follower containers are removed before cold recovery. +The S3 origin remains in that case's Docker volume. +Provider/node failure and unsuccessful drain prevent a case from passing. +An independent-machine kill or power-loss test remains a separate profile. + +Individual reports expose achieved rate, failed delivery gates, window costs +and confirmed commands per selected root. They cannot establish the complete +[proposal qualification](../../docs/write-performance-proposal.md): that also +requires three matched repetitions, A/A variance, publication age/frontier +stability, read guardrails and the overload/recovery experiment. Missing evidence +stays explicitly unverified. Raw artifacts are never committed. +Cellule stability reports fit the debt and oldest-publication age over the last +three one-minute segments. A positive slope fails; missing, late, reset, or +fenced observations cannot pass. Native log frontiers remain observations and +grant no durability or collection authority. diff --git a/scripts/perf/build.py b/scripts/perf/build.py new file mode 100644 index 00000000..b3e6e421 --- /dev/null +++ b/scripts/perf/build.py @@ -0,0 +1,157 @@ +"""Export, adapt and build the paired 96-byte SQL workload outside the checkout.""" +import argparse +import hashlib +import json +import shutil +import io +import tarfile +import subprocess +import fcntl +from pathlib import Path +ROOT = Path(__file__).resolve().parents[2] +RUST = 'rust:1.98.1-bookworm@sha256:93ce27a88655056a51dbdd8f5f2d7ddc071c7b0070fb288a37b5a285fc83971e' +CELLD = 'ghcr.io/denoland/celld@sha256:acb0bc2dca628f0dac7c6b43eea42615a79f4943f316f6df7c1da0b86d52150b' +RUSTFS = 'ghcr.io/rustfs/rustfs@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858' + +def sha(path): + with path.open('rb') as stream: + return hashlib.file_digest(stream, 'sha256').hexdigest() + +def replace(path, before, after, count=1): + text = path.read_text() + if text.count(before) != count: + raise RuntimeError(f'workload adaptation no longer matches {path.name}: {before!r}') + path.write_text(text.replace(before, after)) + +def adapt(source): + sql = source / 'crates/cellule-axum/examples/sql.rs' + replace(sql, 'total_cents INTEGER NOT NULL)', 'total_cents INTEGER NOT NULL, value TEXT NOT NULL)') + replace(sql, ' total_cents: i64,', ' total_cents: i64,\n value: String,', 2) + replace(sql, 'INSERT INTO orders (id, total_cents) VALUES (?1, ?2)', 'INSERT INTO orders (id, total_cents, value) VALUES (?1, ?2, ?3)') + replace(sql, 'SqlValue::Integer(input.total_cents),', 'SqlValue::Integer(input.total_cents),\n SqlValue::Text(input.value),') + replace(sql, 'SELECT total_cents FROM orders', 'SELECT total_cents, value FROM orders', 2) + replace(sql, 'let [SqlValue::Integer(total_cents)]', 'let [SqlValue::Integer(total_cents), SqlValue::Text(value)]') + replace(sql, 'total_cents: *total_cents,', 'total_cents: *total_cents,\n value: value.clone(),') + replace(sql, 'DiskBudget::new(1 << 30)', 'DiskBudget::new(std::env::var("PARITY_DISK_BYTES").unwrap_or_else(|_| "1073741824".into()).parse()?)') + replace(sql, '16 * 1024 * 1024', 'std::env::var("PARITY_RETAINED_BYTES").unwrap_or_else(|_| "67108864".into()).parse()?', 2) + fleet = source / 'crates/cellule-axum/examples/fleet/mod.rs' + replace(fleet, ' let peers = enrollment.authority.recruit().await?;', ' // Fixture policy: keep real node authority for bucket-only runs.\n if std::env::var_os("CELLULE_AXUM_BUCKET_ONLY").is_some() { return Ok(()); }\n let peers = enrollment.authority.recruit().await?;') + replace(fleet, 'println!("Follower durability: enrolled two original boots over pinned mTLS");', + 'if std::env::var_os("CELLULE_AXUM_BUCKET_ONLY").is_some() { println!("Follower durability: bucket-only fixture; node lease installed"); } else { println!("Follower durability: enrolled two original boots over pinned mTLS"); }') + request = source / 'crates/cellule-axum/examples/capacity/request.rs' + replace(request, ' pub total_cents: i64,', ' pub total_cents: i64,\n pub value: String,') + replace(request, ' total_cents: i64,', ' total_cents: i64,\n value: String,') + replace(request, ' id,\n total_cents:', ' id,\n value: "p".repeat(96),\n total_cents:') + replace(request, ' total_cents: body.total_cents,', ' total_cents: body.total_cents,\n value: body.value.clone(),') + +def build(args): + destination = args.artifacts.resolve() + if destination == ROOT or ROOT in destination.parents: + raise RuntimeError('performance artifacts must be outside the repository') + destination.mkdir(parents=True, exist_ok=True) + cache = (args.target_cache or destination / 'target').resolve() + if cache == ROOT or ROOT in cache.parents: + raise RuntimeError('build cache must be outside the repository') + cache.mkdir(parents=True, exist_ok=True) + source = destination / 'source' + if source.exists(): + raise RuntimeError('use a fresh artifact directory; source manifests are immutable') + revision = args.framework_ref or 'HEAD' + resolved = subprocess.check_output(['git', 'rev-parse', revision], cwd=ROOT, text=True).strip() + if args.framework_ref: + source.mkdir() + archive = subprocess.check_output(['git', 'archive', resolved], cwd=ROOT) + with tarfile.open(fileobj=io.BytesIO(archive)) as stream: + stream.extractall(source, filter='data') + else: + files = subprocess.check_output(['git', 'ls-files', '-z', '--cached', '--others', '--exclude-standard'], cwd=ROOT).decode().split('\x00') + for name in files: + if name and (ROOT / name).is_file(): + path = source / name + path.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(ROOT / name, path) + overlay = {} + if args.measurement_overlay: + if not args.framework_ref: + raise RuntimeError('measurement overlay requires a pinned baseline ref') + audited = '397f500a39d0cd3a84724a82fe66d6a56e76f726' + if resolved not in (audited, '18eff0f7af47fac09b993157bb444e582072d7cf'): + raise RuntimeError('the measurement overlay was audited only against the PR 65 foundation') + if subprocess.check_output(['git', 'diff', '--name-only', audited, resolved, '--', + 'crates', 'Cargo.toml', 'Cargo.lock'], cwd=ROOT).strip(): + raise RuntimeError('the merged baseline differs from the audited production source') + # These files contain measurement hooks and the identical driver only. + # The actor files differ from the baseline only in publication timing fields. + # Never overlay codecs, retention, or command execution behavior. + names = [ + 'crates/cellule-store/src/observation/mod.rs', + 'crates/cellule-store/src/observation/tests.rs', + 'crates/cellule-runtime/src/fleet/telemetry.rs', + 'crates/cellule-runtime/src/cell/actor/requests.rs', + 'crates/cellule-runtime/src/cell/actor/mod.rs', + 'crates/cellule-runtime/src/cell/actor/runtime.rs', + 'crates/cellule-runtime/src/cell/actor/state.rs', + 'crates/cellule-runtime/src/cell/actor/task.rs', + 'crates/cellule-runtime/src/cell/actor/tasks/activation.rs', + 'crates/cellule-runtime/src/cell/actor/tasks/publication.rs', + 'crates/cellule-runtime/src/cell/actor/inventory/tests.rs', + 'crates/cellule-runtime/src/lib.rs', + 'crates/cellule-runtime/api-prelude.txt', + 'crates/cellule-runtime/src/node/log/mod.rs', + 'crates/cellule-runtime/src/node/log/tests.rs', + 'crates/cellule-runtime/src/node/durability/mod.rs', + 'crates/cellule-runtime/src/follower/mod.rs', + 'crates/cellule-runtime/src/follower/records/append.rs', + 'crates/cellule-runtime/src/follower/tests/append.rs', + 'crates/cellule-axum/examples/capacity/mod.rs', + 'crates/cellule-axum/examples/sql.rs', + 'crates/cellule-axum/examples/sql_metrics/mod.rs', + 'crates/cellule-axum/examples/sql_metrics/capture.rs', + 'crates/cellule-axum/examples/sql_metrics/storage.rs', + 'crates/cellule-axum/examples/sql_metrics/tests.rs', + 'crates/cellule-axum/examples/fleet/mod.rs', + 'crates/cellule-axum/examples/fleet/authority.rs', + 'crates/cellule-axum/examples/fleet/server.rs', + 'crates/cellule-axum/examples/fleet/transport.rs', + ] + for name in names: + shutil.copy2(ROOT / name, source / name) + overlay[name] = sha(source / name) + manifest = {str(path.relative_to(source)): sha(path) for path in source.rglob('*') if path.is_file()} + (destination / 'framework-source.json').write_text(json.dumps(manifest, sort_keys=True, indent=2) + '\n') + adapt(source) + examples = source / 'crates/cellule-axum/examples' + shutil.copy2(ROOT / 'scripts/perf/http_audit.rs', examples / 'http_audit.rs') + # Cargo uses mtimes for local package freshness. Re-exporting a different + # revision to /work/source can otherwise reuse newer artifacts from the + # previous revision. Namespace the cache by every adapted source byte and + # the pinned compiler image; never rely on export timestamps for identity. + build_source = {str(path.relative_to(source)): sha(path) + for path in sorted(source.rglob('*')) if path.is_file()} + cache_key = hashlib.sha256(json.dumps({'rust': RUST, 'source': build_source}, + sort_keys=True).encode()).hexdigest() + target = cache / cache_key + target.mkdir(exist_ok=True) + for name in ('control.py', 'wait-store-ready.py'): + shutil.copy2(ROOT / 'scripts/perf' / name, destination / name) + shutil.copytree(ROOT / 'scripts/perf/celld', destination / 'celld-app') + subprocess.run(['python3', str(ROOT / 'scripts/generate-capacity-tls.py'), str(destination / 'tls')], check=True) + docker = ['docker', '--context', args.context] + with (target / '.build-lock').open('a') as lock, (destination / 'build.log').open('x') as log: + fcntl.flock(lock, fcntl.LOCK_EX) + subprocess.run(docker + ['run', '--rm', '--cpus', '8', '--memory', '12g', '-v', f'{destination}:/work', '-v', f'{target}:/work/target', '-w', '/work/source', '-e', 'CARGO_TARGET_DIR=/work/target', RUST, 'cargo', 'build', '--release', '--locked', '-p', 'cellule-axum', '--example', 'sql', '--example', 'http_capacity', '--example', 'http_audit'], stdout=log, stderr=subprocess.STDOUT, check=True) + (destination / 'bin').mkdir() + for name in ('sql', 'http_capacity', 'http_audit'): + shutil.copy2(target / 'release/examples' / name, destination / 'bin' / name) + binaries = {name: {'path': f'bin/{name}', 'sha256': sha(destination / f'bin/{name}')} for name in ('sql', 'http_capacity', 'http_audit')} + data = {'schema_version': 2, 'build_source_sha256': cache_key, 'framework_revision': resolved, 'measurement_overlay': overlay, 'framework_source_manifest_sha256': sha(destination / 'framework-source.json'), 'adapted_source': {str(p.relative_to(source)): sha(p) for p in sorted(examples.rglob('*.rs'))}, 'images': {'rust': RUST, 'celld': CELLD, 'store': RUSTFS}, 'celld_revision': 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'workload': 'sql-ledger-96', 'binaries': binaries} + (destination / 'build.json').write_text(json.dumps(data, sort_keys=True, indent=2) + '\n') + print(json.dumps({'manifest': str(destination / 'build.json'), 'binaries': binaries})) +if __name__ == '__main__': + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--context', required=True, help='dedicated Linux Docker context') + parser.add_argument('--artifacts', type=Path, required=True) + parser.add_argument('--framework-ref', help='export a pinned Git revision instead of the working tree') + parser.add_argument('--measurement-overlay', action='store_true', help='apply the current measurement-only files to the pinned baseline') + parser.add_argument('--target-cache', type=Path, help='optional external Linux build cache; binaries are copied into each run') + build(parser.parse_args()) diff --git a/scripts/perf/celld/index.js b/scripts/perf/celld/index.js new file mode 100644 index 00000000..902a626f --- /dev/null +++ b/scripts/perf/celld/index.js @@ -0,0 +1,58 @@ +// Comparison-only application. Framework code and durability gates are unchanged. +export class Orders { + constructor(state) { + this.state = state; + this.sql = state.storage.sql; + this.sql.exec('CREATE TABLE IF NOT EXISTS orders (id INTEGER PRIMARY KEY, total_cents INTEGER NOT NULL, value TEXT NOT NULL)'); + this.sql.exec('CREATE TABLE IF NOT EXISTS commands (request_id TEXT PRIMARY KEY, input TEXT NOT NULL, response TEXT NOT NULL)'); + this.sql.exec('CREATE TABLE IF NOT EXISTS metadata (id INTEGER PRIMARY KEY, incarnation TEXT NOT NULL, sequence INTEGER NOT NULL)'); + this.sql.exec('INSERT OR IGNORE INTO metadata VALUES (1, ?, 0)', crypto.randomUUID().replaceAll('-', '')); + } + receipt() { + const row = this.sql.exec('SELECT incarnation, sequence FROM metadata WHERE id = 1').one(); + return {cell:this.state.id.toString(), incarnation:row.incarnation, commit_sequence:row.sequence}; + } + read(id) { + const rows = this.sql.exec('SELECT id, total_cents, value FROM orders WHERE id = ?', id).toArray(); + return {output:rows[0] ?? null, receipt:this.receipt()}; + } + async fetch(request) { + if (request.method === 'GET') { + const id = Number(new URL(request.url).pathname.split('/').pop()); + return Response.json(this.read(id)); + } + const body = await request.json(); + const input = {id:body.id,total_cents:body.total_cents,value:body.value ?? 'p'.repeat(96)}; + const canonical = JSON.stringify(input); + if (!Number.isSafeInteger(input.id) || !Number.isSafeInteger(input.total_cents) || typeof input.value !== 'string') { + return Response.json({error:'invalid input'}, {status:400}); + } + let result; + let status = 201; + this.state.storage.transactionSync(() => { + const saved = this.sql.exec('SELECT input, response FROM commands WHERE request_id = ?', body.request_id).toArray()[0]; + if (saved) { + if (saved.input !== canonical) {status=409;result={error:'request identity conflict'};return;} + result=JSON.parse(saved.response);return; + } + if (body.expires_at_ms <= Date.now() || body.issued_at_ms > Date.now()) { + status=400;result={error:'request identity expired or not yet valid'};return; + } + this.sql.exec('INSERT INTO orders (id, total_cents, value) VALUES (?, ?, ?)', input.id,input.total_cents,input.value); + this.sql.exec('UPDATE metadata SET sequence = sequence + 1 WHERE id = 1'); + result=this.read(input.id); + this.sql.exec('INSERT INTO commands VALUES (?, ?, ?)', body.request_id,canonical,JSON.stringify(result)); + }); + return Response.json(result, {status}); + } +} +export default { + async fetch(request, env) { + let id; + if (request.method === 'POST') id=(await request.clone().json()).id; + else id=Number(new URL(request.url).pathname.split('/').pop()); + const count=Number(env.CELLS); + const shard=((id % count)+count)%count; + return env.ORDERS.get(env.ORDERS.idFromName(String(shard))).fetch(request); + } +}; diff --git a/scripts/perf/celld/wrangler.json b/scripts/perf/celld/wrangler.json new file mode 100644 index 00000000..0cd77a35 --- /dev/null +++ b/scripts/perf/celld/wrangler.json @@ -0,0 +1,25 @@ +{ + "name": "matched-orders", + "main": "index.js", + "no_bundle": true, + "compatibility_date": "2026-01-01", + "durable_objects": { + "bindings": [ + { + "name": "ORDERS", + "class_name": "Orders" + } + ] + }, + "migrations": [ + { + "tag": "v1", + "new_sqlite_classes": [ + "Orders" + ] + } + ], + "vars": { + "CELLS": "1000" + } +} diff --git a/scripts/perf/compare.py b/scripts/perf/compare.py new file mode 100644 index 00000000..f91b9f0a --- /dev/null +++ b/scripts/perf/compare.py @@ -0,0 +1,97 @@ +"""Compare retained cases without treating overload completions as capacity.""" +import argparse +import hashlib +import json +from pathlib import Path + +from report import case_report, read + + +CONTRACT = ('durability', 'resident_cells', 'concurrency', 'queue_capacity', + 'retained_budget_bytes', 'managed_disk_budget_bytes', 'value_bytes', + 'owner_cpus', 'owner_memory_bytes', 'followers', 'profile', + 'provider_storage', 'seconds', 'warmup_seconds') + + +def summarize(directory): + case = read(directory / 'case.json') + build_path = directory.parent / 'build.json' + build = read(build_path) + report = case_report(directory) + if hashlib.sha256(build_path.read_bytes()).hexdigest() != report['build_manifest_sha256']: + raise ValueError('build manifest changed after the run: ' + str(directory)) + return case, build, report + + +def compare(matrix): + if set(matrix) != {'baseline', 'candidate', 'celld'}: + raise ValueError('matrix needs baseline, candidate and celld case lists') + rows = {} + contract = binaries = images = None + evidence = [] + for role, directories in matrix.items(): + if not directories: + raise ValueError('missing cases for ' + role) + rows[role] = [] + for name in directories: + directory = Path(name).resolve() + case, build, report = summarize(directory) + if case['system'] != ('celld' if role == 'celld' else 'cellule'): + raise ValueError('wrong system for ' + role) + identity = {key: case[key] for key in CONTRACT} + driver = {key: build['binaries'][key]['sha256'] + for key in ('http_capacity', 'http_audit')} + if contract is None: + contract, binaries, images = identity, driver, build['images'] + if identity != contract or driver != binaries or build['images'] != images: + raise ValueError('workload, resources, client or image mismatch: ' + str(directory)) + rows[role].append(report) + evidence.append({'role': role, 'directory': str(directory), + 'source_manifest_sha256': build['framework_source_manifest_sha256'], + 'binary_sha256': build['binaries']['sql']['sha256'], + 'measurement_overlay': bool(build['measurement_overlay']), + 'completed': report['completed'], 'failures': report['failures']}) + points = {} + for role, reports in rows.items(): + by_offer = {} + for report in reports: + for point in report['points']: + key = (point['offered_writes_per_second'], point['offered_reads_per_second']) + by_offer.setdefault(key, []).append(point) + points[role] = by_offer + common = set(points['baseline']) & set(points['candidate']) & set(points['celld']) + if not common: + raise ValueError('no identical offered write/read points') + comparisons = [] + for key in sorted(common): + arms = {} + for role in rows: + samples = points[role][key] + arms[role] = { + 'repetitions': len(samples), + 'rates': [point['successful_writes_per_second'] for point in samples], + 'scheduled_p99_ms': [point['writes']['scheduled_latency_ms_all_attempts']['p99'] for point in samples], + 'delivery_latency_audit_pass': all(point['delivery_latency_audit_pass'] for point in samples), + 'failures': [point['failures'] for point in samples], + 'publication_stability': [point['publication_stability'] for point in samples], + 'provider_cost': [point['window_cost'] for point in samples], + } + comparisons.append({'offered_writes_per_second': key[0], + 'offered_reads_per_second': key[1], 'arms': arms}) + return {'schema_version': 1, 'workload': 'sql-ledger-96', 'contract': contract, + 'evidence': evidence, 'points': comparisons, + 'qualification_pass': False, + 'unverified': ['A/A sustainable-capacity search', 'read-only and mixed capacity guardrails', + 'safe overload refusals and recovery', 'owner loss before materialization', + 'bounded KV workload', 'complete M0–M5 exit gates']} + + +if __name__ == '__main__': + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('matrix', type=Path, help='JSON with three lists of external case directories') + parser.add_argument('--output', type=Path, required=True) + args = parser.parse_args() + result = compare(read(args.matrix)) + args.output.write_text(json.dumps(result, indent=2) + '\n') + print(json.dumps({'report': str(args.output), 'qualification_pass': False, + 'matched_points': len(result['points'])})) diff --git a/scripts/perf/control.py b/scripts/perf/control.py new file mode 100644 index 00000000..cf90d23e --- /dev/null +++ b/scripts/perf/control.py @@ -0,0 +1,42 @@ +import sys, json, urllib.request, urllib.error, datetime, hashlib, hmac + +def request(method, url, body=None): + data = None if body is None else json.dumps(body).encode() + req = urllib.request.Request(url, data=data, method=method, headers={'Content-Type': 'application/json'}) + try: + with urllib.request.urlopen(req, timeout=90) as r: + raw = r.read().decode() + return {'status': r.status, 'body': json.loads(raw) if raw else None} + except urllib.error.HTTPError as e: + raw = e.read().decode() + try: + body = json.loads(raw) + except ValueError: + body = raw + return {'status': e.code, 'body': body} + +def bucket(): + t = datetime.datetime.now(datetime.timezone.utc) + stamp = t.strftime('%Y%m%dT%H%M%SZ') + day = t.strftime('%Y%m%d') + payload = hashlib.sha256(b'').hexdigest() + host = '127.0.0.1:9000' + path = '/comparison' + headers = f'host:{host}\nx-amz-content-sha256:{payload}\nx-amz-date:{stamp}\n' + signed = 'host;x-amz-content-sha256;x-amz-date' + canonical = f'PUT\n{path}\n\n{headers}\n{signed}\n{payload}' + scope = f'{day}/us-east-1/s3/aws4_request' + msg = f'AWS4-HMAC-SHA256\n{stamp}\n{scope}\n' + hashlib.sha256(canonical.encode()).hexdigest() + key = b'AWS4benchmark_secret_private' + for part in [day, 'us-east-1', 's3', 'aws4_request']: + key = hmac.new(key, part.encode(), hashlib.sha256).digest() + sig = hmac.new(key, msg.encode(), hashlib.sha256).hexdigest() + auth = f'AWS4-HMAC-SHA256 Credential=benchmark_access/{scope}, SignedHeaders={signed}, Signature={sig}' + req = urllib.request.Request('http://' + host + path, data=b'', method='PUT', headers={'Authorization': auth, 'x-amz-date': stamp, 'x-amz-content-sha256': payload}) + with urllib.request.urlopen(req) as r: + print(r.status) +if __name__ == '__main__': + if sys.argv[1] == 'bucket': + bucket() + else: + print(json.dumps(request(sys.argv[1], sys.argv[2], json.loads(sys.argv[3]) if len(sys.argv) > 3 else None))) diff --git a/scripts/perf/http_audit.rs b/scripts/perf/http_audit.rs new file mode 100644 index 00000000..daec4ef4 --- /dev/null +++ b/scripts/perf/http_audit.rs @@ -0,0 +1,132 @@ +//! Comparison fixture: validate every acknowledged row and its exact stored retry. +use serde::{Deserialize, Serialize}; +use std::{sync::Arc, time::Duration}; +#[derive(Deserialize)] +struct Config { + address: String, + observations: String, + cold: bool, +} +#[derive(Deserialize, Serialize, PartialEq)] +struct Order { + id: i64, + total_cents: i64, + value: String, +} +#[derive(Deserialize)] +struct Receipt { + cell: String, + incarnation: String, + commit_sequence: u64, +} +#[derive(Deserialize)] +struct Observation { + output: Order, + receipt: Receipt, + #[serde(default)] + request: serde_json::Value, +} +#[derive(Default, Serialize)] +struct Counts { + checked: u64, + retries_checked: u64, + errors: u64, + changed_incarnations: u64, + first_errors: Vec, +} +#[tokio::main(flavor = "multi_thread", worker_threads = 4)] +async fn main() -> Result<(), Box> { + let path = std::env::args() + .nth(1) + .ok_or("configuration path required")?; + let config: Config = serde_json::from_slice(&std::fs::read(path)?)?; + let observations: Vec = + serde_json::from_slice(&std::fs::read(&config.observations)?)?; + if observations.iter().any(|observation| !observation.request.is_object()) { + return Err("audit requires the original request for every acknowledgement".into()); + } + let observations = Arc::new(observations); + let config = Arc::new(config); + let client = reqwest::Client::builder() + .http1_only() + .no_proxy() + .timeout(Duration::from_secs(60)) + .build()?; + let mut tasks = tokio::task::JoinSet::new(); + for index in 0..128 { + let (observations, config, client) = (observations.clone(), config.clone(), client.clone()); + tasks.spawn(async move { + let mut counts = Counts::default(); + for expected in observations.iter().skip(index).step_by(128) { + let result = async { + let actual: Observation = client + .get(format!( + "http://{}/orders/{}", + config.address, expected.output.id + )) + .send() + .await? + .error_for_status()? + .json() + .await?; + if actual.output != expected.output + || actual.receipt.cell != expected.receipt.cell + || actual.receipt.commit_sequence < expected.receipt.commit_sequence + || (!config.cold + && actual.receipt.incarnation != expected.receipt.incarnation) + { + return Err(format!( + "row {} differs from acknowledged output or receipt", + expected.output.id + ) + .into()); + } + let retry: serde_json::Value = client + .post(format!("http://{}/orders", config.address)) + .json(&expected.request) + .send().await?.error_for_status()?.json().await?; + if retry["output"] != serde_json::to_value(&expected.output)? + || retry["receipt"]["cell"].as_str() != Some(expected.receipt.cell.as_str()) + || retry["receipt"]["commit_sequence"].as_u64() != Some(expected.receipt.commit_sequence) + || (!config.cold && retry["receipt"]["incarnation"].as_str() + != Some(expected.receipt.incarnation.as_str())) + { + return Err(format!("row {} stored retry differs", expected.output.id).into()); + } + Ok::<_, Box>( + actual.receipt.incarnation != expected.receipt.incarnation, + ) + } + .await; + counts.checked += 1; + match result { + Ok(changed) => { + counts.changed_incarnations += u64::from(changed); + counts.retries_checked += 1; + } + Err(error) => { + counts.errors += 1; + if counts.first_errors.len() < 4 { + counts.first_errors.push(error.to_string()); + } + } + } + } + counts + }); + } + let mut total = Counts::default(); + while let Some(result) = tasks.join_next().await { + let counts = result?; + total.checked += counts.checked; + total.retries_checked += counts.retries_checked; + total.errors += counts.errors; + total.changed_incarnations += counts.changed_incarnations; + total.first_errors.extend(counts.first_errors); + } + println!("{}", serde_json::to_string(&total)?); + if total.errors > 0 { + return Err("recovery audit failed".into()); + } + Ok(()) +} diff --git a/scripts/perf/report.py b/scripts/perf/report.py new file mode 100644 index 00000000..26f6b626 --- /dev/null +++ b/scripts/perf/report.py @@ -0,0 +1,225 @@ +"""Produce explicit qualification failures, measured rates and window cost deltas.""" +import argparse +import hashlib +import json +from pathlib import Path + +def read(path): + return json.loads(path.read_text()) + +def subtract(before, after): + if isinstance(after, dict): + return {key: subtract(before[key], value) for key, value in after.items() if key in before and (isinstance(value, (dict, list)) or isinstance(value, (int, float)))} + if isinstance(after, list): + if len(before) != len(after): + raise ValueError('metric bucket shape changed') + return [subtract(a, b) for a, b in zip(before, after)] + delta = after - before + if delta < 0: + raise ValueError('monotonic metric reset during the measurement window') + return delta + +def delivery_failures(point, durability): + failures = [] + if point.get('driver_exit_code') != 0: + failures.append('driver failed') + for kind in ('writes', 'reads'): + if kind not in point: + continue + result = point[kind] + if result['planned_offers'] == 0: + continue + prefix = '' if kind == 'writes' else 'reads: ' + for name in ('errors', 'warmup_errors', 'queue_dropped', 'warmup_queue_dropped'): + if result[name] != 0: + failures.append(prefix + name) + if result['generated_offers'] != result['planned_offers']: + failures.append(prefix + 'unissued offers') + if result['successes_in_window'] < result['planned_offers'] * 0.99: + failures.append(prefix + 'less than 99% completed inside window') + p99 = result['scheduled_latency_ms_all_attempts']['p99'] + if p99 is None or (kind == 'writes' and p99 > (50 if durability == 'fleet' else 200)): + failures.append(prefix + 'scheduled p99') + return failures + +def histogram_buckets(histogram): + if 'buckets' in histogram: + return histogram['buckets'] + count = histogram['bucket_count'] + if not isinstance(count, int) or not 2 <= count <= 100002: + raise ValueError('invalid raw histogram bound') + buckets = [0] * count + previous = -1 + for index, value in histogram['nonzero_buckets']: + if not isinstance(index, int) or not previous < index < count or not isinstance(value, int) or value <= 0: + raise ValueError('invalid raw histogram index or count') + buckets[index] = value + previous = index + return buckets + +def histogram_delta(before, after): + if before['resolution_us'] != after['resolution_us']: + raise ValueError('histogram resolution changed') + return {'resolution_us': after['resolution_us'], + 'total_ns': subtract(before['total_ns'], after['total_ns']), + 'buckets': subtract(histogram_buckets(before), histogram_buckets(after))} + +def metric_delta(directory): + start = directory / 'metrics-window-start.json' + end = directory / 'metrics-window-end.json' + if not start.exists() or not end.exists(): + return {'available': False} + before, after = (read(start), read(end)) + if [x['url'] for x in before] != [x['url'] for x in after]: + raise ValueError('metric endpoints changed') + output = [] + for a, b in zip(before, after): + first, last = (a['metrics'], b['metrics']) + if first['schema_version'] != last['schema_version'] or first['sample_session'] != last['sample_session']: + raise ValueError('process or metric schema changed') + if first['histograms'].keys() != last['histograms'].keys(): + raise ValueError('histogram phases changed') + families = subtract(first['storage_families'], last['storage_families']) + counters = lambda snapshot: {operation: {key: value for key, value in fields.items() if key in ('started', 'outcomes', 'bytes_read', 'bytes_written')} for operation, fields in snapshot['storage_operations'].items()} + totals = subtract(counters(first), counters(last)) + for operation, values in totals.items(): + for field in ('started', 'bytes_read', 'bytes_written'): + if field in values and sum((family[operation][field] for family in families.values())) != values[field]: + raise ValueError(f'unclassified {operation}.{field}') + for outcome, count in values.get('outcomes', {}).items(): + if sum((family[operation]['outcomes'][outcome] for family in families.values())) != count: + raise ValueError(f'unclassified {operation}.{outcome}') + output.append({'url': a['url'], 'sample_elapsed_ns': subtract(first['sample_elapsed_ns'], last['sample_elapsed_ns']), 'start_request_ms': [a['request_started_ms'], a['request_finished_ms']], 'end_request_ms': [b['request_started_ms'], b['request_finished_ms']], 'storage_families': families, 'storage': totals, 'histograms': {name: histogram_delta(first['histograms'][name], value) for name, value in last['histograms'].items()}, 'publication': subtract({name: first['writes'][name] for name in ('selected_roots', 'materialized_commits', 'publication_failures', 'uploaded_objects', 'uploaded_bytes')}, {name: last['writes'][name] for name in ('selected_roots', 'materialized_commits', 'publication_failures', 'uploaded_objects', 'uploaded_bytes')}), 'runtime_start': first.get('runtime'), 'runtime_end': last.get('runtime')}) + return {'available': True, 'endpoints': output} + +def cost_report(metrics, successes): + if not metrics.get('available') or successes == 0: + return {'available': False} + endpoints = metrics['endpoints'] + operations = {name: {'attempts': 0, 'successes': 0, 'bytes_read': 0, 'bytes_written': 0} + for name in ('put', 'get', 'range', 'head', 'list', 'copy', 'delete', + 'multipart_start', 'multipart_part', 'multipart_complete', 'multipart_abort')} + families = {} + roots = commits = 0 + for endpoint in endpoints: + publication = endpoint['publication'] + roots += publication['selected_roots'] + commits += publication['materialized_commits'] + for name, total in operations.items(): + source = endpoint['storage'].get(name, {}) + total['attempts'] += source.get('started', 0) + total['successes'] += source.get('outcomes', {}).get('success', 0) + for field in ('bytes_read', 'bytes_written'): + total[field] += source.get(field, 0) + for name, source in endpoint['storage_families'].items(): + family = families.setdefault(name, {'put_attempts': 0, 'put_successes': 0, 'get_attempts': 0, + 'range_get_attempts': 0}) + family['put_attempts'] += source['put']['started'] + family['put_successes'] += source['put']['outcomes']['success'] + family['get_attempts'] += source['get']['started'] + source.get('range', {}).get('started', 0) + family['range_get_attempts'] += source.get('range', {}).get('started', 0) + return {'available': True, 'denominator': 'successful logical commands inside the window', + 'operations': operations, 'families': families, + 'all_provider_put_successes_per_command': operations['put']['successes'] / successes, + 'all_provider_put_attempts_per_command': operations['put']['attempts'] / successes, + 'get_attempts_including_ranges_per_command': (operations['get']['attempts'] + operations['range']['attempts']) / successes, + 'mutation_request_successes_per_command': sum(operations[name]['successes'] for name in ('put', 'copy', 'multipart_start', 'multipart_part', 'multipart_complete', 'multipart_abort')) / successes, + 'observation_scope': 'storage API calls; provider SDK internal retries are not individually instrumented', + 'selected_roots': roots, 'materialized_commits': commits, + 'materialized_commits_per_selected_root': commits / roots if roots else None, + 'trailing_publication_included': False, + 'enrollment_gets_separately_attributed': all('owner_enrollment' in endpoint['storage_families'] and 'receiver_enrollment' in endpoint['storage_families'] for endpoint in endpoints), + 'fresh_enrollment_get_attempts_per_command': sum(families.get(name, {}).get('get_attempts', 0) for name in ('owner_enrollment', 'receiver_enrollment')) / successes if all('owner_enrollment' in endpoint['storage_families'] for endpoint in endpoints) else None} + +def stability_report(directory): + """Fit the last three minute segments; absent/reset evidence cannot pass.""" + paths = [directory / 'metrics-window-start.json', + *sorted(directory.glob('metrics-window-minute-*.json'), + key=lambda path: int(path.stem.rsplit('-', 1)[1])), + directory / 'metrics-window-end.json'] + if any(not path.exists() for path in paths) or len(paths) < 4: + return {'available': False, 'pass': False, 'reason': 'three minute segments missing'} + samples = [read(path) for path in paths[-4:]] + owners = [[item['metrics'] for item in sample + if item['metrics'].get('runtime') is not None] for sample in samples] + if any(len(owner) != 1 for owner in owners): + return {'available': False, 'pass': False, 'reason': 'one owner observation required'} + values = [owner[0] for owner in owners] + if len({(value['sample_session'], value['schema_version']) for value in values}) != 1: + return {'available': False, 'pass': False, 'reason': 'process or schema changed'} + times = [value['sample_elapsed_ns'] / 1e9 for value in values] + if any(b <= a or b - a < 50 or b - a > 70 for a, b in zip(times, times[1:])): + return {'available': False, 'pass': False, 'reason': 'minute observations missing or late'} + if any('error' in value.get('publication_progress', {'error': 'missing'}) for value in values): + return {'available': False, 'pass': False, 'reason': 'publication progress unavailable'} + series = { + 'unpublished_node_log_bytes': [value['runtime']['unpublished_node_log_bytes'] for value in values], + 'oldest_unpublished_ms': [value['publication_progress']['oldest_unpublished_ms'] or 0 for value in values], + 'pending_publications': [value['publication_progress']['pending_publications'] for value in values], + 'retained_capture_bytes': [value['publication_progress']['retained_capture_bytes'] for value in values], + } + center = sum(times) / len(times) + denominator = sum((time - center) ** 2 for time in times) + slopes = {name: sum((time - center) * value for time, value in zip(times, data)) / denominator + for name, data in series.items()} + frontiers = [value.get('node_log_progress') for value in values] + frontier_valid = all(frontier is None or ('error' not in frontier and not frontier['fenced'] + and frontier['tiered_through'] <= frontier['issued_through'] + and frontier['follower_proven_through'] <= frontier['issued_through']) + for frontier in frontiers) + return {'available': True, + 'pass': frontier_valid and slopes['unpublished_node_log_bytes'] <= 0 + and slopes['oldest_unpublished_ms'] <= 0, + 'method': 'least-squares slope over the last three one-minute segments; positive debt or age fails', + 'sample_elapsed_seconds': times, 'series': series, 'slopes_per_second': slopes, + 'frontiers': frontiers, 'frontier_valid': frontier_valid} + +def case_report(directory): + case, summary = (read(directory / 'case.json'), read(directory / 'summary.json')) + failures = [] + exclusions = directory / 'qualification-exclusions.json' + if exclusions.exists(): + failures.append('explicit evidence exclusion: ' + json.dumps(read(exclusions), sort_keys=True)) + if not summary['completed']: + failures.append(summary.get('failure') or 'case incomplete') + for label in ('warm_audit', 'cold_audit'): + audit = summary.get(label, {}) + expected = summary.get('acknowledged_rows') + if expected is None or audit.get('checked') != expected or audit.get('retries_checked') != expected or (audit.get('errors') != 0) or (audit.get('exit_code') != 0): + failures.append(label + ' incomplete or failed') + if not summary.get('cold_retry_pass'): + failures.append('cold contract retry missing') + if summary.get('cleanup_failures'): + failures.append('drain or cleanup failed') + if case.get('telemetry') == 'off': + failures.append('exporter-off diagnostic cannot qualify') + if case.get('diagnostic'): + failures.append('diagnostic profile') + points = [] + for path in sorted(directory.glob('write-*/summary.json')): + config, data = (read(path.parent / 'config.json'), read(path)) + if config.get('phase') in ('overload', 'recovery'): + continue + point_failures = delivery_failures(data, case['durability']) + failures + if config['seconds'] < 300 or config['warmup_seconds'] < 30: + point_failures.append('short diagnostic window') + try: + metrics = metric_delta(path.parent) + except (KeyError, ValueError) as error: + metrics = {'available': False, 'error': str(error)} + point_failures.append('metric evidence invalid') + writes = data['writes'] + stability = stability_report(path.parent) if case['system'] == 'cellule' else {'available': False, 'pass': False, 'reason': 'celld publication age not exposed by this fixture'} + points.append({'offered_writes_per_second': config['write_rate'], 'offered_reads_per_second': config['read_rate'], 'successful_writes_per_second': writes['successful_requests_per_second'], 'logical_value_bytes_per_second': writes['successful_requests_per_second'] * 96, 'writes': writes, 'reads': data['reads'], 'window_metrics': metrics, 'window_cost': cost_report(metrics, writes['successes_in_window']), 'publication_stability': stability, 'delivery_latency_audit_pass': not point_failures, 'failures': point_failures}) + overload = summary.get('overload', {}) + recovery = overload.get('recovery') + recovery_failures = delivery_failures(recovery, case['durability']) if recovery else ['recovery phase missing'] + return {'schema_version': 1, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} +if __name__ == '__main__': + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('directory', type=Path) + args = parser.parse_args() + report = case_report(args.directory) + destination = args.directory / 'report.json' + destination.write_text(json.dumps(report, indent=2) + '\n') + print(json.dumps({'report': str(destination), 'qualification_pass': report['qualification_pass'], 'points': [{key: point[key] for key in ('offered_writes_per_second', 'successful_writes_per_second', 'delivery_latency_audit_pass', 'failures')} for point in report['points']]})) diff --git a/scripts/perf/run.py b/scripts/perf/run.py new file mode 100644 index 00000000..d394f4ce --- /dev/null +++ b/scripts/perf/run.py @@ -0,0 +1,485 @@ +import os, sys, json, subprocess, time, uuid, threading, traceback, hashlib, math, shutil, fcntl, re +from pathlib import Path +BASE = Path(__file__).resolve().parent +CTX = None +RUST = 'rust:1.98.1-bookworm@sha256:93ce27a88655056a51dbdd8f5f2d7ddc071c7b0070fb288a37b5a285fc83971e' +CELLD = 'ghcr.io/denoland/celld@sha256:acb0bc2dca628f0dac7c6b43eea42615a79f4943f316f6df7c1da0b86d52150b' +DOCKER = [] +CONTROL = None +STORE = None +ARGS = None +MANIFEST = None +RUNNER_SHA256 = hashlib.sha256(Path(__file__).read_bytes()).hexdigest() +CREDS = ['-e', 'AWS_ACCESS_KEY_ID=benchmark_access', '-e', 'AWS_SECRET_ACCESS_KEY=benchmark_secret_private', '-e', 'AWS_DEFAULT_REGION=us-east-1'] + +def docker(*args, check=True, timeout=700): + p = subprocess.run(DOCKER + list(map(str, args)), capture_output=True, text=True, timeout=timeout) + if check and p.returncode: + raise RuntimeError(f'docker {args[0]} failed: {p.stdout} {p.stderr}') + return p + +def put(path, obj): + path.write_text(json.dumps(obj, indent=2) + '\n') + +def http(method, path, body=None, port=8080): + args = ['exec', CONTROL, 'python3', '/work/control.py', method, f'http://127.0.0.1:{port}{path}'] + if body is not None: + args.append(json.dumps(body)) + return json.loads(docker(*args, timeout=120).stdout) + +def storage_format_smoke(directory, prefix): + # Check the provider's actual persisted codec before measuring. A source + # manifest alone cannot detect an incorrectly reused Cargo library. + name = 'crates/cellule-ltx/src/replica/root.rs' + source = (BASE / 'source' / name).read_bytes() + if hashlib.sha256(source).hexdigest() != json.loads((BASE / 'framework-source.json').read_text())[name]: + raise RuntimeError('root codec source differs from the immutable manifest') + versions = re.findall(rb'version: ([0-9]+),', source) + if len(versions) != 1: + raise RuntimeError('cannot establish expected root format') + expected = int(versions[0]) + script = r''' +import datetime, hashlib, hmac, json, sys, urllib.request, urllib.parse, xml.etree.ElementTree as ET +def get(path, query=''): + now = datetime.datetime.now(datetime.timezone.utc) + day, stamp = now.strftime('%Y%m%d'), now.strftime('%Y%m%dT%H%M%SZ') + payload = hashlib.sha256(b'').hexdigest() + host = '127.0.0.1:9000' + headers = f'host:{host}\nx-amz-content-sha256:{payload}\nx-amz-date:{stamp}\n' + signed = 'host;x-amz-content-sha256;x-amz-date' + canonical = f'GET\n{path}\n{query}\n{headers}\n{signed}\n{payload}' + scope = f'{day}/us-east-1/s3/aws4_request' + message = f'AWS4-HMAC-SHA256\n{stamp}\n{scope}\n' + hashlib.sha256(canonical.encode()).hexdigest() + key = b'AWS4benchmark_secret_private' + for part in (day, 'us-east-1', 's3', 'aws4_request'): + key = hmac.new(key, part.encode(), hashlib.sha256).digest() + signature = hmac.new(key, message.encode(), hashlib.sha256).hexdigest() + auth = f'AWS4-HMAC-SHA256 Credential=benchmark_access/{scope}, SignedHeaders={signed}, Signature={signature}' + request = urllib.request.Request('http://' + host + path + ('?' + query if query else ''), headers={'Authorization': auth, 'x-amz-date': stamp, 'x-amz-content-sha256': payload}) + with urllib.request.urlopen(request, timeout=30) as response: + return response.read() +cell = sys.argv[3] +cell_prefix = sys.argv[1] + '/cells/v1/apps/' + '03' * 16 + '/cells/' + cell +control = json.loads(get('/comparison/' + cell_prefix + '/control.json')) +objects = cell_prefix + '/inc/' + control['incarnation'] + '/objects/' +root_path = objects + control['root']['digest'] + '.root' +raw = get('/comparison/' + root_path) +root = json.loads(raw) +expected = int(sys.argv[2]) +if root['version'] != expected or root['cell'] != cell or root['incarnation'] != control['incarnation']: + raise SystemExit('persisted root codec or scope differs from source and control') +packed = [segment for segment in root['segments'] if segment.get('packed')] +if expected == 2: + if not packed: + raise SystemExit('small-root fixture did not persist a packed dependency') + body = get('/comparison/' + objects + packed[0]['object_digest'] + '.pack') + if body[:8] != b'CRBPACK1' or len(body) > 256 * 1024: + raise SystemExit('packed dependency has an invalid header or bound') +print(json.dumps({'root_path': root_path, 'root_sha256': hashlib.sha256(raw).hexdigest(), 'version': root['version'], 'packed_segments': len(packed), 'inline_directory': root.get('directory_inline') is not None, 'pass': True})) +''' + cell = json.loads((directory / 'seeds.json').read_text())[0]['receipt']['cell'] + result = docker('exec', CONTROL, 'python3', '-c', script, prefix, expected, cell, timeout=90) + put(directory / 'storage-format-smoke.json', json.loads(result.stdout)) + +def logs(name): + return docker('logs', name, check=False).stdout + docker('logs', name, check=False).stderr + +def wait_ready(name, marker=None, port=8080, limit=600): + started = time.monotonic() + while time.monotonic() - started < limit: + state = json.loads(docker('inspect', name).stdout)[0]['State'] + if not state['Running']: + raise RuntimeError(f'{name} exited: {logs(name)[-12000:]}') + if marker: + if marker in logs(name): + return time.monotonic() - started + else: + try: + if http('GET', '/.well-known/celld/health', port=port)['status'] == 200: + return time.monotonic() - started + except Exception: + pass + time.sleep(1) + raise RuntimeError(f'{name} readiness timeout: {logs(name)[-12000:]}') + +def snapshot(directory, label, names): + data = {} + failures = [] + for name in names + [STORE, CONTROL]: + data[name] = {'inspect': json.loads(docker('inspect', name).stdout)[0]} + p = docker('exec', name, 'sh', '-c', 'cat /proc/1/limits; cat /proc/1/status; ls /proc/1/fd | wc -l; cat /sys/fs/cgroup/memory.events; cat /sys/fs/cgroup/cpu.stat', check=False) + data[name]['process'] = p.stdout + p.stderr + state = data[name]['inspect']['State'] + if state['OOMKilled'] or not state['Running']: + failures.append(f'{name}: {state}') + put(directory / f'{label}-resources.json', data) + if failures: + raise RuntimeError('infrastructure failed: ' + '; '.join(failures)) + +def sampler(directory, stop): + with (directory / 'docker-stats.jsonl').open('w') as f: + while not stop.is_set(): + try: + p = docker('stats', '--no-stream', '--format', '{{json .}}', check=False, timeout=30) + except Exception as error: + f.write(json.dumps({'timestamp': time.time(), 'error': str(error)}) + '\n') + f.flush() + continue + f.write(json.dumps({'timestamp': time.time(), 'stats': [json.loads(x) for x in p.stdout.splitlines() if x]}) + '\n') + f.flush() + stop.wait(3) + +def start_node(system, durability, index, prefix, name): + port = 8080 + index * 10 + args = ['run', '-d', '--name', name, '--network', 'host', '--cpus', '8', '--memory', '16g', '--memory-swap', '16g', '--ulimit', 'nofile=65536:65536', '--tmpfs', '/scratch:rw,size=4g', *CREDS] + if system == 'cellule': + args += ['-e', f'PARITY_RETAINED_BYTES={ARGS.retained_bytes}', '-e', f'PARITY_DISK_BYTES={ARGS.disk_bytes}', '-v', f'{BASE}:/work:ro', '-e', 'TMPDIR=/scratch', '-e', 'CELLULE_AXUM_CELLS=1000', '-e', 'CELLULE_AXUM_WORKERS=8', '-e', f'CELLULE_AXUM_BIND=127.0.0.1:{port}', '-e', 'CELLULE_TEST_ENDPOINT=http://127.0.0.1:9000', '-e', 'CELLULE_TEST_BUCKET=comparison', '-e', f'CELLULE_TEST_PREFIX={prefix}'] + if durability == 'fleet' or index == 0: + args += ['-e', 'CELLULE_AXUM_FLEET_DIR=/scratch/fleet'] + if durability == 'bucket' and index == 0: + args += ['-e', 'CELLULE_AXUM_BUCKET_ONLY=1'] + if index: + args += ['-e', f'CELLULE_AXUM_FOLLOWER={index}'] + args += [RUST, 'sh', '-c', f"mkdir -m 700 /scratch/fleet && cp /work/tls/ca.crt /work/tls/node-{index}.crt /work/tls/node-{index}.key /scratch/fleet/ && exec /work/{MANIFEST['binaries']['sql']['path']}"] + else: + args += ['-e', f'CELLD_PLACEMENT_WEIGHT={(1000000000 if index == 0 else 1)}', '-e', 'CELLD_WATCH=/scratch', '-e', f'CELLD_DURABILITY={durability}', '-e', 'CELLD_SHUTDOWN_TOTAL_MS=600000', '-e', 'CELLD_TOKIO_THREADS=8', '-e', 'CELLD_REBALANCE_INTERVAL_MS=0', '-e', f"CELLD_NODE={name.replace('-cold', '-owner')}", '-e', 'RUST_LOG=warn,celld::node_log=info', CELLD, '--bucket', f's3://comparison/{prefix}', '--endpoint', 'http://127.0.0.1:9000', '--region', 'us-east-1', '--listen', f'127.0.0.1:{port}', '--internal-listen', f'127.0.0.1:{port + 1}', '--advertise', f'127.0.0.1:{port + 1}'] + docker(*args) + return wait_ready(name, 'Follower service:' if index else 'Orders service:') if system == 'cellule' else wait_ready(name, port=port) + +def stop_node(name, directory): + started = time.monotonic() + state = json.loads(docker('inspect', name).stdout)[0]['State'] + if state['Running']: + if 'celld' in name: + port = 8091 if name.endswith('peer-1') else 8101 if name.endswith('peer-2') else 8081 + response = http('POST', '/shutdown?handoff=preserve', port=port) + put(directory / f'{name}-shutdown-response.json', response) + if response['status'] != 200: + raise RuntimeError('celld preserve shutdown was rejected') + else: + docker('kill', '--signal', 'SIGINT', name) + p = docker('wait', name, timeout=120) + (directory / f'{name}-exit.txt').write_text(p.stdout + p.stderr) + (directory / f'{name}.log').write_text(logs(name)) + state = json.loads(docker('inspect', name).stdout)[0]['State'] + put(directory / f'{name}-final-state.json', state) + if state['OOMKilled'] or state['ExitCode'] != 0: + raise RuntimeError(f'{name} shutdown failed: {state}') + return time.monotonic() - started + +def driver(directory, label, config): + configpath = directory / f'{label}-config.json' + put(configpath, config) + p = docker('exec', CONTROL, '/work/' + MANIFEST['binaries']['http_capacity']['path'], f'/work/{configpath.relative_to(BASE)}', check=False, timeout=int(config['seconds'] + config['warmup_seconds'] + 300)) + (directory / f'{label}-driver.log').write_text(p.stdout + p.stderr) + dest = directory / label + copy = docker('cp', f'{CONTROL}:/tmp/comparison-evidence', dest, check=False, timeout=300) + if copy.returncode: + raise RuntimeError(f'no capacity evidence: {copy.stderr}; {p.stdout} {p.stderr}') + docker('exec', CONTROL, 'rm', '-rf', '/tmp/comparison-evidence') + result = json.loads((dest / 'summary.json').read_text()) + result['driver_exit_code'] = p.returncode + put(dest / 'summary.json', result) + return result + +def config(directory, label, write_rate, read_rate, offset=0, seed=True, seconds=30): + return {'address': '127.0.0.1:8080', 'cells': 1000, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'write_rate': write_rate, 'read_rate': read_rate, 'warmup_seconds': ARGS.warmup if seed else 1, 'seconds': ARGS.seconds if seed else seconds, 'evidence_directory': '/tmp/comparison-evidence', 'seed_file': f'/work/{directory.relative_to(BASE)}/seeds.json' if seed else None, 'write_offset': offset, 'metrics_urls': [], 'metrics_tls_directory': None} + +def contract(directory): + now = int(time.time() * 1000) + body = {'request_id': str(uuid.uuid4()), 'issued_at_ms': now, 'expires_at_ms': now + 7200000, 'id': 1000000000, 'total_cents': 13000000099, 'value': 'p' * 96} + first = http('POST', '/orders', body) + retry = http('POST', '/orders', body) + conflict = http('POST', '/orders', dict(body, total_cents=1)) + expired = dict(body, request_id=str(uuid.uuid4()), id=1000001000, issued_at_ms=now - 10000, expires_at_ms=now - 1000) + reject = http('POST', '/orders', expired) + missing = http('GET', f"/orders/{expired['id']}") + result = {'request': body, 'first': first, 'retry': retry, 'conflict': conflict, 'expired': reject, 'expired_read': missing} + put(directory / 'contract.json', result) + if first['status'] != 201 or retry != first or 200 <= conflict['status'] < 300 or (200 <= reject['status'] < 300) or (missing['body']['output'] is not None): + raise RuntimeError(f'contract failed: {result}') + return dict(first['body'], request=body) + +def collect_observations(directory, extra): + observations = {extra['output']['id']: extra} + for path in [directory / 'initialize' / 'setup.jsonl', *sorted(directory.glob('write-*/client-*.jsonl'))]: + for line in path.open(): + row = json.loads(line) + if row.get('error') is None and row.get('response') and row.get('request'): + response = row['response'] + oid = response['output']['id'] + if oid in observations: + raise RuntimeError(f'nonunique acknowledged id {oid}') + observations[oid] = dict(response, request=row['request']) + put(directory / 'acknowledged.json', list(observations.values())) + return len(observations) + +def audit(directory, label, cold): + path = directory / f'{label}-config.json' + put(path, {'address': '127.0.0.1:8080', 'observations': f'/work/{directory.relative_to(BASE)}/acknowledged.json', 'cold': cold}) + p = docker('exec', CONTROL, '/work/' + MANIFEST['binaries']['http_audit']['path'], f'/work/{path.relative_to(BASE)}', check=False, timeout=1200) + (directory / f'{label}.log').write_text(p.stdout + p.stderr) + try: + result = json.loads(p.stdout.strip().splitlines()[-1]) + except Exception: + result = {'error': p.stdout + p.stderr} + result['exit_code'] = p.returncode + put(directory / f'{label}.json', result) + if p.returncode: + raise RuntimeError(f'{label} failed: {result}') + return result + +def run_case(system, durability): + label = f'{system}-{durability}-parity-' + ARGS.tag + directory = BASE / label + directory.mkdir(exist_ok=False) + prefix = label + '-' + uuid.uuid4().hex[:12] + put(directory / 'case.json', {'system': system, 'durability': durability, 'prefix': prefix, 'framework_commit': MANIFEST['framework_revision'] if system == 'cellule' else 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'resident_cells': 1000, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'candidate_binary': MANIFEST['binaries']['sql']['path'], 'celld_application': 'celld-app', 'retained_budget_bytes': int(ARGS.retained_bytes), 'managed_disk_budget_bytes': int(ARGS.disk_bytes), 'value_bytes': 96, 'owner_cpus': 8, 'owner_memory_bytes': 16 * 1024 ** 3, 'followers': 2 if durability == 'fleet' else 0, 'profile': 'shared-vm-sql-ledger-96', 'provider_storage': 'fresh Linux Docker volume', 'telemetry': ARGS.telemetry, 'warmup_seconds': ARGS.warmup, 'seconds': ARGS.seconds, 'runner_sha256': RUNNER_SHA256, 'diagnostic': ARGS.seconds < 300 or ARGS.warmup < 30 or ARGS.telemetry == 'off'}) + names = [] + stop = threading.Event() + sampling = None + summary = {'schema_version': 1, 'case': label, 'writes': [], 'reads': [], 'completed': False, 'failure': None, 'build_manifest_sha256': hashlib.sha256((BASE / 'build.json').read_bytes()).hexdigest()} + try: + docker('restart', CONTROL) + start_provider(directory / 'store-data') + create_bucket() + for attempt in range(60): + try: + if http('GET', '/health', port=9000)['status'] == 200: + break + except Exception: + pass + time.sleep(1) + ready = docker('exec', CONTROL, 'python3', '/work/wait-store-ready.py', timeout=70) + put(directory / 'store-readiness.json', json.loads(ready.stdout)) + if system == 'celld': + p = docker('run', '--rm', '--network', 'host', '--cpus', '2', '--memory', '2g', '--memory-swap', '2g', '--ulimit', 'nofile=65536:65536', '-v', f"{BASE}/{'celld-app'}:/app:ro", *CREDS, CELLD, 'deploy', '/app', '--bucket', f's3://comparison/{prefix}', '--endpoint', 'http://127.0.0.1:9000', '--region', 'us-east-1') + (directory / 'deploy.log').write_text(p.stdout + p.stderr) + elif durability == 'fleet': + for i in [1, 2]: + name = f'comparison-{label}-peer-{i}' + names.append(name) + start_node(system, durability, i, prefix, name) + if system == 'celld' and durability == 'fleet': + for i in [1, 2]: + name = f'comparison-{label}-peer-{i}' + names.append(name) + start_node(system, durability, i, prefix, name) + owner = f'comparison-{label}-owner' + names.insert(0, owner) + summary['startup_seconds'] = start_node(system, durability, 0, prefix, owner) + sampling = threading.Thread(target=sampler, args=(directory, stop), daemon=True) + sampling.start() + initialize = driver(directory, 'initialize', config(directory, 'initialize', 0, 1, seed=False, seconds=1)) + summary['initialize'] = initialize + seeds = [] + for line in (directory / 'initialize' / 'setup.jsonl').open(): + seeds.append(json.loads(line)['response']) + put(directory / 'seeds.json', seeds) + if system == 'cellule': + storage_format_smoke(directory, prefix) + if system == 'celld' and durability == 'fleet': + end = time.monotonic() + 60 + while time.monotonic() < end: + if any(('log ensemble open; fleet acks enabled' in line and names[1] in line and (names[2] in line) for line in logs(owner).splitlines())): + break + time.sleep(1) + (directory / 'ensemble-ready.log').write_text(logs(owner)) + if not any(('log ensemble open; fleet acks enabled' in line and names[1] in line and (names[2] in line) for line in logs(owner).splitlines())): + raise RuntimeError('celld fleet ensemble did not form with both selected followers') + extra = contract(directory) + if system == 'celld': + states = {str(i): http('GET', '/state', port=8081 + i * 10) for i in range(3 if durability == 'fleet' else 1)} + put(directory / 'placement-before.json', states) + if states['0']['body']['owned_cells'] != 1000 or any((states[str(i)]['body']['owned_cells'] for i in range(1, 3 if durability == 'fleet' else 1))): + raise RuntimeError('celld placement is not one owner of 1000 Cells') + snapshot(directory, 'before', names) + for n, rate in enumerate(ARGS.write_rates or [30, 2000 if durability == 'bucket' else 15000]): + point = config(directory, f'write-{rate}', rate, ARGS.read_rate, offset=n * 100000000) + if system == 'cellule' and ARGS.telemetry == 'window': + point['metrics_urls'] = ['http://127.0.0.1:8080/debug/metrics'] + ([f'https://127.0.0.1:{port}/debug/metrics' for port in (8090, 8100)] if durability == 'fleet' else []) + point['metrics_tls_directory'] = '/work/tls' + point['hot_read_cells'] = ARGS.hot_read_cells + r = driver(directory, f'write-{rate}', point) + summary['writes'].append(r) + put(directory / 'progress.json', summary) + w = r['writes'] + p99 = w['scheduled_latency_ms_all_attempts']['p99'] + slo = 50 if durability == 'fleet' else 200 + passed = r['driver_exit_code'] == 0 and w['errors'] == 0 and (w['warmup_errors'] == 0) and (w['generated_offers'] == w['planned_offers']) and (w['queue_dropped'] == 0) and (w['warmup_queue_dropped'] == 0) and (w['successes_in_window'] >= w['planned_offers'] * 0.99) and (p99 is not None) and (p99 <= slo) + r['delivery_latency_pass'] = passed + put(directory / 'progress.json', summary) + if ARGS.overload_capacity: + # No intervening warmup: the recovery window starts immediately + # after the offered overload. Audit journals include both cohorts. + phases = [('overload', math.ceil(ARGS.overload_capacity * 1.5), 60), + ('recovery', max(1, ARGS.overload_capacity // 2), 30)] + summary['overload'] = {'reference_capacity': ARGS.overload_capacity, + 'reference_requires_paired_qualification': True} + for index, (phase, rate, seconds) in enumerate(phases): + point = config(directory, phase, rate, 0, + offset=(100 + index) * 100000000, seconds=seconds) + point['seconds'] = seconds + point['warmup_seconds'] = 0 + point['phase'] = phase + r = driver(directory, 'write-' + phase, point) + summary['overload'][phase] = r + put(directory / 'progress.json', summary) + summary['acknowledged_rows'] = collect_observations(directory, extra) + summary['warm_audit'] = audit(directory, 'warm-audit', False) + snapshot(directory, 'after', names) + if system == 'celld': + put(directory / 'placement-after.json', {str(i): http('GET', '/state', port=8081 + i * 10) for i in range(3 if durability == 'fleet' else 1)}) + stop.set() + sampling.join(timeout=40) + drain_started = time.monotonic() + summary['drain_node_seconds'] = {name: stop_node(name, directory) for name in names} + summary['drain_seconds'] = time.monotonic() - drain_started + if summary['drain_seconds'] > 120: + raise RuntimeError('original fleet drain exceeded 120 seconds') + for name in names: + docker('rm', name) + names = [] + owner = f'comparison-{label}-cold' + names.append(owner) + summary['cold_startup_seconds'] = start_node(system, 'bucket', 0, prefix, owner) + summary['cold_audit'] = audit(directory, 'cold-audit', True) + saved = json.loads((directory / 'contract.json').read_text()) + retry = http('POST', '/orders', saved['request']) + put(directory / 'cold-retry.json', retry) + if retry['status'] != 201 or retry['body']['output'] != saved['first']['body']['output']: + raise RuntimeError('cold retry did not preserve acknowledged output') + summary['cold_retry_pass'] = True + stop_node(owner, directory) + docker('rm', owner) + names = [] + summary['completed'] = True + except Exception as e: + summary['completed'] = False + summary['failure'] = str(e)[:2000] + (directory / 'failure.txt').write_text(traceback.format_exc()) + print(f'FAILED {label}: {str(e)[:500]}', flush=True) + finally: + stop.set() + if sampling: + sampling.join(timeout=40) + for name in names: + try: + stop_node(name, directory) + except Exception as e: + summary.setdefault('cleanup_failures', []).append(str(e)) + (directory / f'{name}.log').write_text(logs(name)) + docker('stop', '--time', '2', name, check=False) + (directory / 'store.log').write_text(logs(STORE)) + put(directory / 'summary.json', summary) + print(json.dumps({'case': label, 'completed': summary['completed'], 'failure': str(summary.get('failure'))[:500], 'write_rates': [x['writes']['successful_requests_per_second'] for x in summary['writes']], 'read_rates': [x.get('successful_requests_per_second') for x in summary['reads']]}), flush=True) + return summary + +def remove_owned(name): + previous = docker('inspect', name, check=False) + if previous.returncode == 0: + value = json.loads(previous.stdout)[0] + if value['Config'].get('Labels', {}).get('cellule.perf.artifacts') != str(BASE): + raise RuntimeError(f'{name} belongs to another run') + docker('rm', '-f', name) + + +def start_provider(data_directory): + # A fresh prefix does not reset a provider's old object inventory. Preserve + # each case's origin data outside the repository and reset the provider's + # backing directory before the next case, including A/A repetitions. + remove_owned(STORE) + data_directory.mkdir(parents=True, exist_ok=False) + volume = STORE + '-data-' + hashlib.sha256(str(data_directory).encode()).hexdigest()[:12] + if docker('volume', 'inspect', volume, check=False).returncode == 0: + raise RuntimeError('provider volume is not fresh: ' + volume) + docker('volume', 'create', '--label', 'cellule.perf.artifacts=' + str(BASE), volume) + put(data_directory / 'volume.json', {'name': volume, 'storage': 'Docker Linux filesystem', 'retained': True}) + docker('run', '-d', '--name', STORE, '--label', 'cellule.perf.artifacts=' + str(BASE), + '--network', 'host', '--cpus', '2', '--memory', '2g', '--memory-swap', '2g', + '--ulimit', 'nofile=65536:65536', '-e', 'RUSTFS_ACCESS_KEY=benchmark_access', + '-e', 'RUSTFS_SECRET_KEY=benchmark_secret_private', + '-v', f'{volume}:/data', MANIFEST['images']['store'], '/data') + + +def create_bucket(): + for _ in range(60): + try: + result = docker('exec', CONTROL, 'python3', '/work/control.py', 'bucket', check=False, timeout=10) + if result.returncode == 0: + return + except Exception: + pass + time.sleep(1) + raise RuntimeError('fixture bucket creation failed') + + +def provision(): + remove_owned(CONTROL) + docker('run', '-d', '--name', CONTROL, '--label', 'cellule.perf.artifacts=' + str(BASE), + '--network', 'host', '--cpus', '4', '--memory', '4g', '-v', f'{BASE}:/work', + RUST, 'sleep', 'infinity') +if __name__ == '__main__': + import argparse + parser = argparse.ArgumentParser(description='Paired, pinned SQL durability verification; raw evidence stays outside the repository.') + parser.add_argument('--context', required=True) + parser.add_argument('--artifacts', type=Path, required=True) + parser.add_argument('--tag', default=uuid.uuid4().hex[:10]) + parser.add_argument('--seconds', type=int, default=300) + parser.add_argument('--warmup', type=int, default=30) + parser.add_argument('--repetitions', type=int, default=3) + parser.add_argument('--write-rates', type=int, nargs='+') + parser.add_argument('--read-rate', type=int, default=0) + parser.add_argument('--hot-read-cells', type=int) + parser.add_argument('--telemetry', choices=['window', 'off'], default='window', + help='off is an exporter-overhead diagnostic and cannot qualify') + parser.add_argument('--concurrency', type=int, default=128) + parser.add_argument('--queue-capacity', type=int, default=128) + parser.add_argument('--retained-bytes', type=int, default=67108864) + parser.add_argument('--disk-bytes', type=int, default=1073741824) + parser.add_argument('--overload-capacity', type=int, + help='after steady points offer 1.5x this qualified capacity for 60s, then 0.5x for 30s') + parser.add_argument('cases', nargs='*', choices=['cellule-fleet', 'celld-fleet', 'cellule-bucket', 'celld-bucket']) + ARGS = parser.parse_args() + BASE = ARGS.artifacts.resolve() + CTX = ARGS.context + if BASE == Path(__file__).resolve().parents[2] or Path(__file__).resolve().parents[2] in BASE.parents: + parser.error('artifacts must stay outside the repository') + if not 1 <= ARGS.repetitions <= 10 or not 1 <= ARGS.seconds <= 3600 or (not 1 <= ARGS.warmup <= 60): + parser.error('invalid bounded duration/repetition') + if ARGS.overload_capacity is not None and ARGS.overload_capacity <= 0: + parser.error('overload capacity must be positive') + DOCKER = ['docker', '--context', CTX] + MANIFEST = json.loads((BASE / 'build.json').read_text()) + if MANIFEST.get('schema_version') != 2 or not MANIFEST.get('build_source_sha256'): + parser.error('rebuild with source-content-isolated caches before running this harness') + for binary in MANIFEST['binaries'].values(): + if hashlib.sha256((BASE / binary['path']).read_bytes()).hexdigest() != binary['sha256']: + parser.error('binary hash differs from build manifest') + prefix = 'cellule-perf-' + hashlib.sha256(str(BASE).encode()).hexdigest()[:10] + CONTROL = prefix + '-control' + STORE = prefix + '-store' + lock_path = Path('/tmp') / ('cellule-perf-' + hashlib.sha256(CTX.encode()).hexdigest()[:16] + '.lock') + benchmark_lock = lock_path.open('a') + try: + fcntl.flock(benchmark_lock, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError: + raise SystemExit('Another comparison owns this Docker context') + provision() + original_tag = ARGS.tag + results = [] + try: + for repeat in range(ARGS.repetitions): + ARGS.tag = original_tag + '-r' + str(repeat + 1) + cases = ARGS.cases or ['cellule-fleet', 'celld-fleet', 'cellule-bucket', 'celld-bucket'] + if repeat % 2: + cases = list(reversed(cases)) + for spec in cases: + system, durability = spec.split('-') + results.append(run_case(system, durability)) + finally: + for name in (CONTROL, STORE): + docker('stop', '--time', '10', name, check=False) + put(BASE / (original_tag + '-runs.json'), results) + if any((not result['completed'] for result in results)): + raise SystemExit(1) diff --git a/scripts/perf/wait-store-ready.py b/scripts/perf/wait-store-ready.py new file mode 100644 index 00000000..07840bfa --- /dev/null +++ b/scripts/perf/wait-store-ready.py @@ -0,0 +1,35 @@ +import datetime, hashlib, hmac, json, time, urllib.request, urllib.error +started = time.monotonic() +stable = 0 +attempts = 0 +while time.monotonic() - started < 60: + attempts += 1 + now = datetime.datetime.now(datetime.timezone.utc) + date = now.strftime('%Y%m%d') + stamp = now.strftime('%Y%m%dT%H%M%SZ') + payload = hashlib.sha256(b'').hexdigest() + query = 'list-type=2&max-keys=1' + host = '127.0.0.1:9000' + headers = 'host:' + host + '\nx-amz-content-sha256:' + payload + '\nx-amz-date:' + stamp + '\n' + signed = 'host;x-amz-content-sha256;x-amz-date' + canonical = 'GET\n/comparison\n' + query + '\n' + headers + '\n' + signed + '\n' + payload + scope = date + '/us-east-1/s3/aws4_request' + message = 'AWS4-HMAC-SHA256\n' + stamp + '\n' + scope + '\n' + hashlib.sha256(canonical.encode()).hexdigest() + + def mac(key, value): + return hmac.new(key, value.encode(), hashlib.sha256).digest() + key = mac(mac(mac(mac(b'AWS4benchmark_secret_private', date), 'us-east-1'), 's3'), 'aws4_request') + signature = hmac.new(key, message.encode(), hashlib.sha256).hexdigest() + request = urllib.request.Request('http://' + host + '/comparison?' + query, headers={'x-amz-date': stamp, 'x-amz-content-sha256': payload, 'Authorization': 'AWS4-HMAC-SHA256 Credential=benchmark_access/' + scope + ', SignedHeaders=' + signed + ', Signature=' + signature}) + try: + with urllib.request.urlopen(request, timeout=5) as response: + body = response.read() + ready = response.status == 200 and b'ListBucketResult' in body + stable = stable + 1 if ready else 0 + except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError): + stable = 0 + if stable >= 3: + print(json.dumps({'bucket_list_ready': True, 'attempts': attempts, 'seconds': round(time.monotonic() - started, 3)})) + raise SystemExit(0) + time.sleep(1) +raise SystemExit('bucket list readiness timed out') diff --git a/scripts/tests/test_perf_report.py b/scripts/tests/test_perf_report.py new file mode 100644 index 00000000..df94a52e --- /dev/null +++ b/scripts/tests/test_perf_report.py @@ -0,0 +1,135 @@ +"""High completion rates and fast failed responses cannot pass write gates.""" +import importlib.util +import unittest +import json +import tempfile +from pathlib import Path + +SPEC = importlib.util.spec_from_file_location("perf_report", Path(__file__).parents[1] / "perf/report.py") +REPORT = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(REPORT) + + +class DeliveryGateTests(unittest.TestCase): + def point(self): + return {"driver_exit_code": 0, "writes": { + "errors": 0, "warmup_errors": 0, "queue_dropped": 0, "warmup_queue_dropped": 0, + "generated_offers": 4500000, "planned_offers": 4500000, + "successes_in_window": 4500000, "scheduled_latency_ms_all_attempts": {"p99": 10}}} + + def test_fast_error_responses_fail_even_when_latency_passes(self): + point = self.point() + point["writes"]["errors"] = 100 + self.assertIn("errors", REPORT.delivery_failures(point, "fleet")) + + def test_dropped_or_unissued_offers_cannot_be_hidden_by_successful_tps(self): + for field in ("queue_dropped", "warmup_queue_dropped"): + point = self.point() + point["writes"][field] = 1 + self.assertIn(field, REPORT.delivery_failures(point, "fleet")) + point = self.point() + point["writes"]["generated_offers"] -= 1 + self.assertIn("unissued offers", REPORT.delivery_failures(point, "fleet")) + + def test_cumulative_histogram_reset_refuses_a_window(self): + with self.assertRaises(ValueError): + REPORT.subtract({"buckets": [3, 2]}, {"buckets": [3, 1]}) + + def test_resolution_change_cannot_be_interpreted_as_a_latency_improvement(self): + before = {'resolution_us': 100, 'total_ns': 100000, 'buckets': [0, 1, 0]} + after = {'resolution_us': 10, 'total_ns': 200000, 'buckets': [0, 2, 0]} + with self.assertRaisesRegex(ValueError, 'resolution changed'): + REPORT.histogram_delta(before, after) + + def test_sparse_histogram_preserves_new_buckets_and_rejects_resets(self): + before = {'bucket_count': 5, 'nonzero_buckets': [[1, 3], [4, 1]]} + after = {'bucket_count': 5, 'nonzero_buckets': [[1, 4], [2, 2], [4, 1]]} + self.assertEqual(REPORT.subtract(REPORT.histogram_buckets(before), + REPORT.histogram_buckets(after)), [0, 1, 2, 0, 0]) + with self.assertRaises(ValueError): + REPORT.histogram_buckets({'bucket_count': 5, 'nonzero_buckets': [[1, 3], [1, 2]]}) + with self.assertRaises(ValueError): + REPORT.subtract(REPORT.histogram_buckets(before), + REPORT.histogram_buckets({'bucket_count': 5, 'nonzero_buckets': [[4, 1]]})) + + def test_read_errors_fail_a_mixed_point(self): + point = self.point() + point["reads"] = dict(point["writes"], errors=1) + self.assertIn("reads: errors", REPORT.delivery_failures(point, "fleet")) + + def test_range_reads_and_multipart_work_are_not_hidden_by_full_get_put_counts(self): + def operation(count): + return {'started': count, 'outcomes': {'success': count}, + 'bytes_read': 0, 'bytes_written': 0} + metrics = {'available': True, 'endpoints': [{ + 'publication': {'selected_roots': 2, 'materialized_commits': 2}, + 'storage': {'get': operation(2), 'range': operation(3), 'put': operation(4), + 'multipart_part': operation(3)}, + 'storage_families': {'immutable': {'get': operation(2), 'range': operation(3), + 'put': operation(4)}}}]} + cost = REPORT.cost_report(metrics, 2) + self.assertEqual(cost['get_attempts_including_ranges_per_command'], 2.5) + self.assertEqual(cost['mutation_request_successes_per_command'], 3.5) + self.assertEqual(cost['families']['immutable']['get_attempts'], 5) + + def test_positive_publication_age_or_debt_trend_cannot_pass_stability(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + names = ['metrics-window-start.json', 'metrics-window-minute-1.json', + 'metrics-window-minute-2.json', 'metrics-window-end.json'] + for index, name in enumerate(names): + sample = [{'metrics': {'schema_version': 2, 'sample_session': 'owner', + 'sample_elapsed_ns': index * 60 * 10**9, + 'runtime': {'unpublished_node_log_bytes': 1000 + index}, + 'publication_progress': {'oldest_unpublished_ms': 10, + 'pending_publications': 1, 'retained_capture_bytes': 1000}}}] + (directory / name).write_text(json.dumps(sample)) + result = REPORT.stability_report(directory) + self.assertTrue(result['available']) + self.assertFalse(result['pass']) + self.assertGreater(result['slopes_per_second']['unpublished_node_log_bytes'], 0) + for name in names: + path = directory / name + sample = json.loads(path.read_text()) + sample[0]['metrics']['runtime']['unpublished_node_log_bytes'] = 0 + sample[0]['metrics']['publication_progress']['oldest_unpublished_ms'] = None + path.write_text(json.dumps(sample)) + self.assertTrue(REPORT.stability_report(directory)['pass']) + (directory / names[1]).unlink() + self.assertFalse(REPORT.stability_report(directory)['pass']) + + def test_window_reconciles_actual_storage_schema_and_preserves_resolution(self): + def sample(count): + operation = {"started": count, "outcomes": {"success": count}, + "bytes_read": 0, "bytes_written": count * 64} + return [{"url": "http://127.0.0.1:8080/debug/metrics", + "request_started_ms": count, "request_finished_ms": count + 1, + "metrics": {"schema_version": 1, "sample_session": "fixed", + "sample_elapsed_ns": count * 1000, + "storage_families": {"immutable": {"put": operation}}, + "storage_operations": {"put": operation}, + "histograms": {"worker": {"resolution_us": 100, + "total_ns": count * 200000, "buckets": [0, 0, count]}}, + "writes": {"selected_roots": count, + "materialized_commits": count * 2, + "publication_failures": 0, + "uploaded_objects": count, "uploaded_bytes": count * 64}}}] + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + (directory / "metrics-window-start.json").write_text(json.dumps(sample(1))) + (directory / "metrics-window-end.json").write_text(json.dumps(sample(5))) + window = REPORT.metric_delta(directory) + endpoint = window["endpoints"][0] + self.assertEqual(endpoint["histograms"]["worker"]["resolution_us"], 100) + self.assertEqual(endpoint["histograms"]["worker"]["buckets"], [0, 0, 4]) + self.assertEqual(endpoint["publication"]["materialized_commits"], 8) + bad = sample(5) + bad[0]["metrics"]["storage_operations"]["put"] = dict( + bad[0]["metrics"]["storage_operations"]["put"], bytes_written=999) + (directory / "metrics-window-end.json").write_text(json.dumps(bad)) + with self.assertRaisesRegex(ValueError, "unclassified"): + REPORT.metric_delta(directory) + + +if __name__ == "__main__": + unittest.main() From 409c1643f261435a83d28b728e908efe2c65a9ae Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 6 Oct 2026 20:36:37 -0700 Subject: [PATCH 2/8] Record paired write evidence and preserve frozen qualification fixtures --- .../qualification/routing.py | 25 ++++- .../qualification/test_routing.py | 11 +++ docs/write-performance-delivery.md | 77 +++++++++++---- scripts/perf/compare.py | 37 ++++++- scripts/perf/report.py | 4 +- scripts/tests/test_perf_compare.py | 99 +++++++++++++++++++ 6 files changed, 227 insertions(+), 26 deletions(-) create mode 100644 scripts/tests/test_perf_compare.py diff --git a/crates/cellule-peer-http/qualification/routing.py b/crates/cellule-peer-http/qualification/routing.py index 1a01cbbb..80898209 100644 --- a/crates/cellule-peer-http/qualification/routing.py +++ b/crates/cellule-peer-http/qualification/routing.py @@ -56,6 +56,24 @@ def output(*command, cwd=None): return subprocess.check_output(command, cwd=cwd, text=True).strip() +def adapt_test_harness(harness, telemetry): + """Keep synthetic unit-test initializers valid on the frozen baseline. + + Measurement callbacks use the common timing fields. These two extra fields + are only in unit-test fixtures, whose coverage is reconstructed from the + same commit sequences on both revisions. + """ + if "pub covered_commits: u64" in telemetry: + return harness + text = harness.decode() + for line in (" covered_commits: if sequence == 1 { 1 } else { 4 },\n", + " covered_commits: 1,\n"): + if text.count(line) != 1: + raise RuntimeError("Frozen routing harness initializer no longer matches") + text = text.replace(line, "") + return text.encode() + + def build(revision, name, state, evidence, harness): source = state / f"{name}-source" source.mkdir() @@ -63,7 +81,9 @@ def build(revision, name, state, evidence, harness): with tarfile.open(fileobj=io.BytesIO(archive)) as files: files.extractall(source, filter="data") # Only test wiring is transplanted. Older revisions predate this harness. - (source / PEER / "src/performance_tests.rs").write_bytes(harness) + adapted_harness = adapt_test_harness( + harness, (source / "crates/cellule-runtime/src/fleet/telemetry.rs").read_text()) + (source / PEER / "src/performance_tests.rs").write_bytes(adapted_harness) entry = source / PEER / "src/lib.rs" if "mod performance_tests;" not in entry.read_text(): entry.write_text(entry.read_text() + "\n#[cfg(test)]\nmod performance_tests;\n") @@ -111,7 +131,8 @@ def test_dependencies(match): raise RuntimeError(f"Missing exact benchmark: {listed}") return {"revision": revision, "binary": str(binary), "binary_sha256": digest(binary), "source_archive_sha256": hashlib.sha256(archive).hexdigest(), - "harness_sha256": hashlib.sha256(harness).hexdigest()} + "harness_sha256": hashlib.sha256(harness).hexdigest(), + "adapted_harness_sha256": hashlib.sha256(adapted_harness).hexdigest()} def schedule(order): diff --git a/crates/cellule-peer-http/qualification/test_routing.py b/crates/cellule-peer-http/qualification/test_routing.py index 4f531b12..23a84aff 100644 --- a/crates/cellule-peer-http/qualification/test_routing.py +++ b/crates/cellule-peer-http/qualification/test_routing.py @@ -14,6 +14,17 @@ def measurements(unleased_reads=4096): class GateTests(unittest.TestCase): + def test_frozen_harness_changes_only_synthetic_timing_initializers(self): + harness = (routing.Path(__file__).parents[1] / "src/performance_tests.rs").read_bytes() + self.assertEqual(routing.adapt_test_harness(harness, "pub covered_commits: u64"), harness) + adapted = routing.adapt_test_harness(harness, "pub commit_sequence: u64") + expected = harness.decode().replace( + " covered_commits: if sequence == 1 { 1 } else { 4 },\n", "").replace( + " covered_commits: 1,\n", "") + self.assertEqual(adapted.decode(), expected) + with self.assertRaisesRegex(RuntimeError, "initializer no longer matches"): + routing.adapt_test_harness(adapted, "pub commit_sequence: u64") + @staticmethod def publication(first, last): return dict(first_sequence=first, sequence=last, queue_wait_ns=0, diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 26b65905..e9ffe901 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -36,7 +36,7 @@ collection paths. There is no legacy decoding or automatic migration. | M2 | Existing native grouping and per-Cell coalescing preserved | Shared node publication coordinator not implemented; 0.25 publication PUTs/command not achieved | | M3 | Fresh enrollment roles, peer phases and follower append measured | Signed append grants and durable grant fences not implemented; 0.05 enrollment GETs/command not achieved | | M4 | Existing authority-pinned Cell roots remain the object proof | Bundle coverage proof, transfer and collection protocol not implemented | -| M5 | All-ACK warm/cold audits and five-minute main characterization completed | Paired repetitions, read/failure/overload matrix and absolute/relative parity unverified | +| M5 | One matched five-minute Fleet point and all-ACK warm/cold audits completed | Three repetitions, read/failure/overload matrix and absolute/relative parity unverified | ## Evidence @@ -53,24 +53,63 @@ main files are design documents. The baseline explicitly overlays measurement hooks. Celld is v0.6.1, `f2bf648663a610eefde71f3547ad61e9b896b1f0`, using the pinned container digest. Both framework arms use byte-identical clients/auditors. -Latest-main Fleet offered 100 writes/s for 300 seconds after 30 seconds warmup. -It completed **99.993 writes/s**, with **868.1 ms scheduled p99**, zero errors -and zero drops. Every one of its **34,001 acknowledged commands** passed GET -and exact-command retry audits while warm and after all original node containers -were removed. The original fleet drained in **7.737 seconds**. This point fails -the 50-ms Fleet latency gate and publication stability. - -Its window measured **6.688 successful PUTs/command**, **3.327 GET attempts/command -including ranges**, and **1.256 fresh enrollment GETs/command**, summed over -owner and followers. Unpublished bytes and oldest publication age had positive -slopes over the final three minute segments. The provider was at its two-CPU -ceiling during much of the run. Fleet proof wait averaged 88.0 ms, response -confirmation 0.290 ms, capture 1.047 ms, and tmpfs follower data sync about -0.0007 ms. This points to publication and provider/peer pressure in this profile; -it does not establish the bottleneck for NVMe or managed object storage. - -Candidate and matched celld results will be recorded after their complete audits. -No candidate speedup is inferred from the main result. +Each arm offered 100 Fleet writes/s for 300 seconds after 30 seconds warmup, +serially with a fresh provider volume. These are individual points, not a +capacity search or three-repetition qualification. + +| Window or audit | Latest main + telemetry | Packed candidate | celld | +| --- | ---: | ---: | ---: | +| Successful commands/s inside window | 99.993 | 99.990 | 100.000 | +| Scheduled p50 / p95 / p99, ms | 66.8 / 265.3 / 868.1 | 24.0 / 200.1 / 386.5 | 5.6 / 11.8 / 44.3 | +| Errors / dropped offers | 0 / 0 | 0 / 0 | 0 / 0 | +| Commands checked by GET and exact retry, warm and cold | 34,001 each | 34,001 each | 34,001 each | +| Original fleet drain, seconds | 7.737 | 10.918 | 11.742 | +| All successful storage API PUTs/command | 6.688 | 4.992 | Not instrumented | +| All GET attempts/command, including ranges | 3.327 | 4.059 | Not instrumented | +| Fresh enrollment GETs/command, owner + receivers | 1.256 | 2.347 | Different authorization protocol | +| Logical commits per selected command root | 1.000 | 1.000 | Not instrumented | +| Final debt slope, bytes/s | +2,455.9 | +2,386.2 | Not instrumented | +| Final oldest-publication age slope, ms/s | +3.711 | +4.435 | Not instrumented | +| 50-ms delivery latency gate | Fail | Fail | Pass at this point | + +All original node containers were removed before bucket-only cold audits. This +checks recovery after successful drain, not owner loss before materialization. +The candidate lowered p99 by 55.5% and PUTs/command by 25.4% in this one pair, +but its p99 remains **8.72 times celld's**. Both Cellule arms fail the proposal's +publication stability gate. No sustainable-rate improvement is established. + +Compaction mean fell from 444.8 to 25.6 ms; publication mean from 1,288.7 to +489.4 ms; Fleet proof wait mean from 88.0 to 52.0 ms. Capture remained about +1.1 ms and tmpfs follower data sync about 0.0005–0.0007 ms. The faster candidate +also performed more compactions and enrollment GETs. Reduced batching is a +possible explanation for the enrollment increase; this run does not prove it. +The provider reached its two-CPU ceiling. Publication amplification and fresh +peer work remain priorities in this profile; device sync performance is untested. + +The measured one-command-per-root result cannot satisfy M2's authority-write +budget by sharing data alone. With three per-Cell selection PUTs, the floor is +three PUTs/command before shared data, node coverage or compaction. M3's measured +2.347 enrollment GETs/command is 46.9 times its 0.05 budget. Meeting those gates +requires the specified coalescing/proof/grant protocols and their failure tests. + +| Frozen artifact | Run/build identity | SQL binary SHA-256 prefix | +| --- | --- | --- | +| Latest main + measurement overlay | `implementation-main-logical-metrics` | `85d30c0f93df` | +| Candidate | `implementation-m1-one-fetch` | `ff4c6ede25f4` | +| celld | v0.6.1 container digest pinned in `build.json` | Container digest | + +The client and auditor hashes are respectively `cc1d47522078` and +`35368139e4c5` for both builds. Full digests and every exported source hash live +in the retained manifests. All candidate production bytes match PR 66's +`a1c48fcd`; the final test expectation and documentation were edited after the +binary export. A Git base revision alone does not identify an overlaid build. + +The isolated all-feature workspace suite passed **1,837 tests** with **38 ignored** +environment-dependent tests. Clippy and API documentation passed with warnings +denied; format, boundaries, layout, Rust fences, links and SQL/peer contract +checks passed. The first compaction run exposed an old range-GET expectation; +the corrected test now requires zero range GETs and two complete pack GETs. +That failure remains in the external evidence. ## Reproduce and inspect diff --git a/scripts/perf/compare.py b/scripts/perf/compare.py index f91b9f0a..6f45bea8 100644 --- a/scripts/perf/compare.py +++ b/scripts/perf/compare.py @@ -23,6 +23,27 @@ def summarize(directory): return case, build, report +def read_guardrail(baseline, candidate): + """A matched point is not evidence of either arm's highest read capacity.""" + if not baseline or len(baseline) != len(candidate): + return {'available': False, 'pass': False, 'reason': 'matched repeats missing'} + results = [] + for before, after in zip(baseline, candidate): + rate = before['successful_reads_per_second'] + p99 = before['reads']['scheduled_latency_ms_all_attempts']['p99'] + new_p99 = after['reads']['scheduled_latency_ms_all_attempts']['p99'] + valid = rate > 0 and p99 is not None and p99 > 0 and new_p99 is not None + ratio_rate = after['successful_reads_per_second'] / rate if valid else None + ratio_p99 = new_p99 / p99 if valid else None + passed = bool(valid and before['delivery_latency_audit_pass'] + and after['delivery_latency_audit_pass'] + and ratio_rate >= 0.90 and ratio_p99 <= 1.20) + results.append({'pass': passed, 'throughput_ratio': ratio_rate, + 'scheduled_p99_ratio': ratio_p99}) + return {'available': True, 'pass': all(result['pass'] for result in results), + 'repetitions': results, 'baseline_capacity_search_verified': False} + + def compare(matrix): if set(matrix) != {'baseline', 'candidate', 'celld'}: raise ValueError('matrix needs baseline, candidate and celld case lists') @@ -59,25 +80,33 @@ def compare(matrix): key = (point['offered_writes_per_second'], point['offered_reads_per_second']) by_offer.setdefault(key, []).append(point) points[role] = by_offer - common = set(points['baseline']) & set(points['candidate']) & set(points['celld']) - if not common: - raise ValueError('no identical offered write/read points') + common = set(points['baseline']) + if not common or any(set(arm) != common for arm in points.values()): + raise ValueError('offered point sets differ; no case may be silently omitted') comparisons = [] for key in sorted(common): arms = {} + contracts = [sample['load_contract'] for role in rows for sample in points[role][key]] + if any(contract != contracts[0] for contract in contracts): + raise ValueError('point workload or hot-read distribution mismatch: ' + str(key)) for role in rows: samples = points[role][key] arms[role] = { 'repetitions': len(samples), 'rates': [point['successful_writes_per_second'] for point in samples], 'scheduled_p99_ms': [point['writes']['scheduled_latency_ms_all_attempts']['p99'] for point in samples], + 'read_rates': [point['successful_reads_per_second'] for point in samples], + 'read_scheduled_p99_ms': [point['reads']['scheduled_latency_ms_all_attempts']['p99'] for point in samples], 'delivery_latency_audit_pass': all(point['delivery_latency_audit_pass'] for point in samples), 'failures': [point['failures'] for point in samples], 'publication_stability': [point['publication_stability'] for point in samples], 'provider_cost': [point['window_cost'] for point in samples], } comparisons.append({'offered_writes_per_second': key[0], - 'offered_reads_per_second': key[1], 'arms': arms}) + 'offered_reads_per_second': key[1], 'load_contract': contracts[0], + 'read_guardrail_at_matched_point': read_guardrail( + points['baseline'][key], points['candidate'][key]) if key[1] else None, + 'arms': arms}) return {'schema_version': 1, 'workload': 'sql-ledger-96', 'contract': contract, 'evidence': evidence, 'points': comparisons, 'qualification_pass': False, diff --git a/scripts/perf/report.py b/scripts/perf/report.py index 26f6b626..43b80d20 100644 --- a/scripts/perf/report.py +++ b/scripts/perf/report.py @@ -210,7 +210,9 @@ def case_report(directory): point_failures.append('metric evidence invalid') writes = data['writes'] stability = stability_report(path.parent) if case['system'] == 'cellule' else {'available': False, 'pass': False, 'reason': 'celld publication age not exposed by this fixture'} - points.append({'offered_writes_per_second': config['write_rate'], 'offered_reads_per_second': config['read_rate'], 'successful_writes_per_second': writes['successful_requests_per_second'], 'logical_value_bytes_per_second': writes['successful_requests_per_second'] * 96, 'writes': writes, 'reads': data['reads'], 'window_metrics': metrics, 'window_cost': cost_report(metrics, writes['successes_in_window']), 'publication_stability': stability, 'delivery_latency_audit_pass': not point_failures, 'failures': point_failures}) + load_contract = {key: config.get(key) for key in ('cells', 'concurrency', 'queue_capacity', + 'write_offset', 'seconds', 'warmup_seconds', 'hot_read_cells')} + points.append({'offered_writes_per_second': config['write_rate'], 'offered_reads_per_second': config['read_rate'], 'successful_writes_per_second': writes['successful_requests_per_second'], 'successful_reads_per_second': data['reads']['successful_requests_per_second'], 'load_contract': load_contract, 'logical_value_bytes_per_second': writes['successful_requests_per_second'] * 96, 'writes': writes, 'reads': data['reads'], 'window_metrics': metrics, 'window_cost': cost_report(metrics, writes['successes_in_window']), 'publication_stability': stability, 'delivery_latency_audit_pass': not point_failures, 'failures': point_failures}) overload = summary.get('overload', {}) recovery = overload.get('recovery') recovery_failures = delivery_failures(recovery, case['durability']) if recovery else ['recovery phase missing'] diff --git a/scripts/tests/test_perf_compare.py b/scripts/tests/test_perf_compare.py new file mode 100644 index 00000000..bf39f7ae --- /dev/null +++ b/scripts/tests/test_perf_compare.py @@ -0,0 +1,99 @@ +"""A comparison must retain failed points and reject mismatched workloads.""" +import copy +import sys +import unittest +from pathlib import Path +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).parents[1] / 'perf')) +import compare as comparison + + +def point(): + return {'offered_writes_per_second': 0, 'offered_reads_per_second': 10000, + 'successful_writes_per_second': 0, 'successful_reads_per_second': 10000, + 'writes': {'scheduled_latency_ms_all_attempts': {'p99': None}}, + 'reads': {'scheduled_latency_ms_all_attempts': {'p99': 2}}, + 'load_contract': {'hot_read_cells': None, 'write_offset': 0}, + 'delivery_latency_audit_pass': True, 'failures': [], + 'publication_stability': {'pass': True}, 'window_cost': {'available': False}} + + +class ComparisonTests(unittest.TestCase): + def setUp(self): + self.matrix = {role: [role] for role in ('baseline', 'candidate', 'celld')} + self.arms = {} + for role in self.matrix: + case = dict.fromkeys(comparison.CONTRACT, 'same') + case['system'] = 'celld' if role == 'celld' else 'cellule' + build = {'binaries': {name: {'sha256': name} for name in ('sql', 'http_capacity', 'http_audit')}, + 'images': {'store': 'pinned'}, 'framework_source_manifest_sha256': role, + 'measurement_overlay': role == 'baseline'} + report = {'points': [point()], 'completed': True, 'failures': []} + self.arms[role] = (case, build, report) + + def compare(self): + with patch.object(comparison, 'summarize', side_effect=lambda path: self.arms[path.name]): + return comparison.compare(self.matrix) + + def test_identical_read_point_does_not_claim_capacity_or_full_qualification(self): + result = self.compare() + guardrail = result['points'][0]['read_guardrail_at_matched_point'] + self.assertTrue(guardrail['pass']) + self.assertFalse(guardrail['baseline_capacity_search_verified']) + self.assertFalse(result['qualification_pass']) + self.assertEqual(result['points'][0]['arms']['candidate']['read_rates'], [10000]) + + def test_read_errors_latency_and_throughput_regressions_fail(self): + baseline = point() + for rate, p99, delivered in ((8999, 2, True), (10000, 2.401, True), (10000, 1, False)): + with self.subTest(rate=rate, p99=p99, delivered=delivered): + candidate = point() + candidate['successful_reads_per_second'] = rate + candidate['reads']['scheduled_latency_ms_all_attempts']['p99'] = p99 + candidate['delivery_latency_audit_pass'] = delivered + self.assertFalse(comparison.read_guardrail([baseline], [candidate])['pass']) + self.assertFalse(comparison.read_guardrail([baseline], [])['pass']) + + def test_hot_distribution_and_preceding_load_must_match(self): + for key, value in (('hot_read_cells', 10), ('write_offset', 100000000)): + with self.subTest(key=key): + self.setUp() + self.arms['candidate'][2]['points'][0]['load_contract'][key] = value + with self.assertRaisesRegex(ValueError, 'workload or hot-read distribution mismatch'): + self.compare() + + def test_different_client_resources_or_provider_cannot_compare(self): + for kind in ('client', 'resources', 'image'): + with self.subTest(kind=kind): + self.setUp() + case, build, _ = self.arms['candidate'] + if kind == 'client': + build['binaries']['http_capacity']['sha256'] = 'different' + elif kind == 'resources': + case['owner_cpus'] = 4 + else: + build['images']['store'] = 'different' + with self.assertRaisesRegex(ValueError, 'mismatch'): + self.compare() + + def test_unmatched_offered_points_cannot_disappear_from_comparison(self): + extra = copy.deepcopy(point()) + extra['offered_reads_per_second'] = 20000 + self.arms['candidate'][2]['points'].append(extra) + with self.assertRaisesRegex(ValueError, 'no case may be silently omitted'): + self.compare() + + def test_failed_case_and_point_remain_visible(self): + report = self.arms['candidate'][2] + report['completed'] = False + report['failures'] = ['cold audit failed'] + report['points'][0]['delivery_latency_audit_pass'] = False + report['points'][0]['failures'] = ['cold audit failed'] + result = self.compare() + self.assertIn('cold audit failed', result['evidence'][1]['failures']) + self.assertFalse(result['points'][0]['read_guardrail_at_matched_point']['pass']) + + +if __name__ == '__main__': + unittest.main() From 5ac24b3ae30abc9fffda7976e6135114acc27c41 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 6 Oct 2026 21:14:03 -0700 Subject: [PATCH 3/8] Preserve scheduled offers and bound full acknowledgement audits Keep late arrivals in the original offered cohort, support immediate overload recovery, and stream every successful command through bounded collection and GET/retry audits. Require client, fixture, runner and host provenance for new comparisons. Retain failed read saturation results and regression evidence. --- .../examples/capacity/arrivals/mod.rs | 63 ++++++ .../examples/capacity/arrivals/tests.rs | 77 +++++++ crates/cellule-axum/examples/capacity/mod.rs | 56 ++--- docs/write-performance-delivery.md | 41 ++++ scripts/perf/README.md | 33 +++ scripts/perf/build.py | 8 +- scripts/perf/compare.py | 22 +- scripts/perf/http_audit.rs | 214 ++++++++++++------ scripts/perf/observations.py | 64 ++++++ scripts/perf/report.py | 25 +- scripts/perf/run.py | 42 ++-- scripts/tests/test_perf_compare.py | 39 +++- scripts/tests/test_perf_observations.py | 73 ++++++ scripts/tests/test_perf_report.py | 11 + 14 files changed, 635 insertions(+), 133 deletions(-) create mode 100644 crates/cellule-axum/examples/capacity/arrivals/mod.rs create mode 100644 crates/cellule-axum/examples/capacity/arrivals/tests.rs create mode 100644 scripts/perf/observations.py create mode 100644 scripts/tests/test_perf_observations.py diff --git a/crates/cellule-axum/examples/capacity/arrivals/mod.rs b/crates/cellule-axum/examples/capacity/arrivals/mod.rs new file mode 100644 index 00000000..682939b1 --- /dev/null +++ b/crates/cellule-axum/examples/capacity/arrivals/mod.rs @@ -0,0 +1,63 @@ +//! Emit the complete offered schedule, including offers whose wakeup is late. +use super::{Job, Kind}; +use std::time::Duration; +use tokio::{sync::mpsc, time::Instant}; + +#[cfg(test)] +mod tests; + +#[derive(Default)] +pub(super) struct Arrivals { + pub offered: [u64; 2], + pub dropped: [u64; 2], + pub warmup_dropped: [u64; 2], +} + +pub(super) async fn produce( + sender: mpsc::Sender, + start: Instant, + warm_end: Instant, + end: Instant, + rates: [u64; 2], +) -> Arrivals { + let mut counts = Arrivals::default(); + let mut indexes = [0_u64; 2]; + loop { + let write_due = (indexes[0] * 1_000_000_000) + .checked_div(rates[0]) + .map_or(end, |nanos| start + Duration::from_nanos(nanos)); + let read_due = (indexes[1] * 1_000_000_000) + .checked_div(rates[1]) + .map_or(end, |nanos| start + Duration::from_nanos(nanos)); + let kind = usize::from(read_due < write_due); + let due = if kind == 0 { write_due } else { read_due }; + if due >= end { + break; + } + // Wakeup delay cannot erase an offer scheduled inside the window. + // Its original due time still controls latency and completion gates; + // a full queue records a drop rather than hiding work as unissued. + tokio::time::sleep_until(due).await; + let measured = due >= warm_end; + counts.offered[kind] += u64::from(measured); + let job = Job { + kind: if kind == 0 { Kind::Write } else { Kind::Read }, + index: indexes[kind], + due, + measured, + }; + match sender.try_send(job) { + Ok(()) => {} + Err(mpsc::error::TrySendError::Full(_)) => { + if measured { + counts.dropped[kind] += 1; + } else { + counts.warmup_dropped[kind] += 1; + } + } + Err(mpsc::error::TrySendError::Closed(_)) => break, + } + indexes[kind] += 1; + } + counts +} diff --git a/crates/cellule-axum/examples/capacity/arrivals/tests.rs b/crates/cellule-axum/examples/capacity/arrivals/tests.rs new file mode 100644 index 00000000..336db71b --- /dev/null +++ b/crates/cellule-axum/examples/capacity/arrivals/tests.rs @@ -0,0 +1,77 @@ +use super::*; + +#[tokio::test] +async fn delayed_producer_keeps_every_original_due_time_and_window_offer() { + let start = Instant::now() - Duration::from_secs(5); + let warm_end = start + Duration::from_secs(1); + let end = warm_end + Duration::from_secs(1); + let (sender, mut receiver) = mpsc::channel(20); + let arrivals = produce(sender, start, warm_end, end, [2, 3]).await; + assert_eq!(arrivals.offered, [2, 3]); + assert_eq!(arrivals.dropped, [0, 0]); + assert_eq!(arrivals.warmup_dropped, [0, 0]); + let mut counts = [0; 2]; + let mut measured = [0; 2]; + while let Some(job) = receiver.recv().await { + let kind = match job.kind { + Kind::Write => 0, + Kind::Read => 1, + }; + let rate = [2, 3][kind]; + assert_eq!(job.index, counts[kind]); + assert_eq!( + job.due, + start + Duration::from_nanos(job.index * 1_000_000_000 / rate) + ); + assert!(job.due < end); + assert_eq!(job.measured, job.due >= warm_end); + counts[kind] += 1; + measured[kind] += u64::from(job.measured); + } + assert_eq!(counts, [4, 6]); + assert_eq!(measured, arrivals.offered); +} + +#[tokio::test] +async fn late_full_queue_records_drops_instead_of_omitting_offers() { + let start = Instant::now() - Duration::from_secs(5); + let warm_end = start + Duration::from_secs(1); + let end = warm_end + Duration::from_secs(1); + let (sender, mut receiver) = mpsc::channel(1); + let arrivals = produce(sender, start, warm_end, end, [0, 3]).await; + assert_eq!(arrivals.offered, [0, 3]); + assert_eq!(arrivals.dropped, [0, 3]); + assert_eq!(arrivals.warmup_dropped, [0, 2]); + assert!(!receiver.recv().await.unwrap().measured); + assert!(receiver.recv().await.is_none()); +} + +#[tokio::test] +async fn immediate_recovery_window_needs_no_new_warmup() { + let start = Instant::now() - Duration::from_secs(5); + let end = start + Duration::from_secs(1); + let (sender, mut receiver) = mpsc::channel(4); + let arrivals = produce(sender, start, start, end, [2, 0]).await; + assert_eq!(arrivals.offered, [2, 0]); + assert_eq!(arrivals.warmup_dropped, [0, 0]); + while let Some(job) = receiver.recv().await { + assert!(job.measured); + } +} + +#[test] +fn recovery_metadata_and_zero_warmup_are_supported_by_the_real_config_decoder() { + let config = serde_json::json!({"address":"127.0.0.1:8080", "cells":1000, + "concurrency":128, "queue_capacity":128, "write_rate":10, "read_rate":0, + "warmup_seconds":0, "seconds":30, "evidence_directory":"/tmp/audit", + "phase":"recovery"}); + let decoded: super::super::Config = serde_json::from_value(config.clone()).unwrap(); + decoded.validate().unwrap(); + let mut warmed = config.clone(); + warmed["warmup_seconds"] = serde_json::json!(30); + let decoded: super::super::Config = serde_json::from_value(warmed).unwrap(); + assert!(decoded.validate().is_err()); + let mut unknown = config; + unknown["phase"] = serde_json::json!("hide_steady_point"); + assert!(serde_json::from_value::(unknown).is_err()); +} diff --git a/crates/cellule-axum/examples/capacity/mod.rs b/crates/cellule-axum/examples/capacity/mod.rs index bb62d123..536e7791 100644 --- a/crates/cellule-axum/examples/capacity/mod.rs +++ b/crates/cellule-axum/examples/capacity/mod.rs @@ -8,6 +8,7 @@ use tokio::{ time::Instant, }; +mod arrivals; mod metrics; mod request; use metrics::Metrics; @@ -34,6 +35,14 @@ pub struct Config { metrics_urls: Vec, hot_read_cells: Option, metrics_tls_directory: Option, + phase: Option, +} + +#[derive(Deserialize, Serialize)] +#[serde(rename_all = "lowercase")] +enum ProbePhase { + Overload, + Recovery, } impl Config { @@ -45,7 +54,8 @@ impl Config { || self.write_rate > 100_000 || self.read_rate > 100_000 || self.write_rate + self.read_rate == 0 - || !(1..=60).contains(&self.warmup_seconds) + || self.warmup_seconds > 60 + || (self.phase.is_some() && self.warmup_seconds != 0) || !(1..=3_600).contains(&self.seconds) || !self.write_offset.is_multiple_of(self.cells as u64) || self @@ -376,46 +386,12 @@ pub async fn run(config: Config) -> Result<()> { Ok::<_, Box>((writes, reads)) }); } - let mut indexes = [0_u64; 2]; - let mut offered = [0_u64; 2]; - let mut dropped = [0_u64; 2]; - let mut warmup_dropped = [0_u64; 2]; let rates = [config.write_rate, config.read_rate]; - loop { - let write_due = (indexes[0] * 1_000_000_000) - .checked_div(rates[0]) - .map_or(end, |nanos| start + Duration::from_nanos(nanos)); - let read_due = (indexes[1] * 1_000_000_000) - .checked_div(rates[1]) - .map_or(end, |nanos| start + Duration::from_nanos(nanos)); - let kind = usize::from(read_due < write_due); - let due = if kind == 0 { write_due } else { read_due }; - if due >= end || Instant::now() >= end { - break; - } - tokio::time::sleep_until(due).await; - let measured = due >= warm_end; - offered[kind] += u64::from(measured); - let job = Job { - kind: if kind == 0 { Kind::Write } else { Kind::Read }, - index: indexes[kind], - due, - measured, - }; - match sender.try_send(job) { - Ok(()) => {} - Err(mpsc::error::TrySendError::Full(_)) => { - if measured { - dropped[kind] += 1 - } else { - warmup_dropped[kind] += 1 - } - } - Err(mpsc::error::TrySendError::Closed(_)) => break, - } - indexes[kind] += 1; - } - drop(sender); + let arrivals::Arrivals { + offered, + dropped, + warmup_dropped, + } = arrivals::produce(sender, start, warm_end, end, rates).await; let mut writes = Metrics::new(config.cells); let mut reads = Metrics::new(config.cells); let mut task_errors = Vec::new(); diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index e9ffe901..86ed56e8 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -15,6 +15,8 @@ the acceptance contract; completing tests or a load run does not pass its gates. | Window telemetry separates response, proof and publication | Logical commands per selected root; capture/checkpoint, worker, peer and sync histograms | Counters and frontiers confer no authority or proof | | Storage families distinguish owner/receiver enrollment | Summed enrollment GET cost across all three nodes | Each signed peer message still uses fresh authorization | | Docker runner and reports preserve failures | Source/binary identities, fresh provider volumes, scheduled-arrival latency, all-ACK audits | Errors, drops, unissued offers, provider failures and failed drain cannot pass | +| Arrival producer preserves delayed offers | Final wakeup cannot erase a request scheduled inside the window | Original arrival time still determines latency and completion; full queues count drops | +| ACK collection and audit stream bounded records | Disk-backed uniqueness index; 256 queued records and at most 128 GET/retry pairs | Seed, warmup, steady, trailing, overload and recovery successes all reconcile | An ordinary small root needs two immutable PUTs plus lineage and fenced Cell selection: **four successful PUTs instead of six**. A small scheduled @@ -92,6 +94,40 @@ three PUTs/command before shared data, node coverage or compaction. M3's measure 2.347 enrollment GETs/command is 46.9 times its 0.05 budget. Meeting those gates requires the specified coalescing/proof/grant protocols and their failure tests. +### Read saturation evidence + +The same frozen clients offered 10,000 reads/s for five minutes after 30 seconds +warmup, without writes beyond setup. Every arm passed warm and cold GET/retry +audits of its 1,001 seed/contract mutations, with zero HTTP errors. Every arm +failed delivery qualification through dropped offers. These completion rates +are saturation observations, not qualified read capacities or M1's read guardrail. + +| Window | Latest main + telemetry | Packed candidate | celld | +| --- | ---: | ---: | ---: | +| Successful reads/s inside window | 9,893.81 | 9,934.98 | 9,321.13 | +| Scheduled p50 / p95 / p99, ms | 1.6 / 2.9 / 11.1 | 1.6 / 2.8 / 8.2 | 2.1 / 26.6 / 73.4 | +| Measured queue drops | 31,834 | 19,494 | 203,645 | +| Warmup queue drops | 14,240 | 902 | 12,138 | +| Unissued measured offers | 4 | 10 | 4 | +| Original fleet drain, seconds | 24.235 | 23.321 | 10.491 | + +Inspection and regression tests reproduced why the producer omitted those final +offers: its wall-clock stop could precede emission of an arrival already due +inside the window. The new producer emits the full scheduled cohort and keeps +lateness in the original arrival time. This fixes accounting; it cannot erase +the real queue drops in these historical runs. Three tests failed against the +old guard and passed after its removal. The updated client also accepts the +explicit zero-warmup overload/recovery phases used by the runner. + +The new auditor reads JSONL through bounded queues and checks every original +command, rather than retaining a multi-million-response vector. Collection +rejects duplicate IDs, missing successful journal records and partial input. +The runner verifies stream hashes before and after both audits and records its +loaded source and Docker host limits. New comparisons require matching fixture, +runner, host and client identities; older cases remain marked as lacking that +complete provenance. Rebuild both arms before comparing the revised driver; +historical and new clients are not interchangeable. + | Frozen artifact | Run/build identity | SQL binary SHA-256 prefix | | --- | --- | --- | | Latest main + measurement overlay | `implementation-main-logical-metrics` | `85d30c0f93df` | @@ -110,6 +146,11 @@ denied; format, boundaries, layout, Rust fences, links and SQL/peer contract checks passed. The first compaction run exposed an old range-GET expectation; the corrected test now requires zero range GETs and two complete pack GETs. That failure remains in the external evidence. +The revised client/auditor passed 18 targeted Rust tests and warnings-denied +Clippy; local LTX without replica features passed 58 tests including its doctest. +The Python comparison, report and collector checks passed 21 tests with warnings +treated as errors. These checks do not substitute for live overload, fault or +capacity qualification. ## Reproduce and inspect diff --git a/scripts/perf/README.md b/scripts/perf/README.md index e1d80508..280d2d15 100644 --- a/scripts/perf/README.md +++ b/scripts/perf/README.md @@ -94,6 +94,12 @@ only to measure exporter overhead; that diagnostic cannot qualify. Compare response, confirmation, fleet-proof, capture/checkpoint, and publication timings separately instead of attributing publication time to the command ACK. +The producer emits every offer scheduled inside the window even if its final +wakeup is late. It preserves the original due time: lateness remains in the +scheduled latency and inside-window completion counts, and a full queue records +a drop. Overload and immediate recovery use explicit bounded phase metadata +with zero warmup; ordinary qualification still requires 30 seconds warmup. + Warm and bucket-only cold audits GET every acknowledged mutation and retry **every original command**, checking its exact stored output and sequence. Original owner and follower containers are removed before cold recovery. @@ -101,6 +107,18 @@ The S3 origin remains in that case's Docker volume. Provider/node failure and unsuccessful drain prevent a case from passing. An independent-machine kill or power-loss test remains a separate profile. +New builds use `acknowledged.jsonl`, streamed by a 256-record producer queue +and at most 128 concurrent GET/retry pairs. Collection checks ID uniqueness +with a disposable disk index and reconciles seed, warmup, steady, trailing, +overload and recovery ACK counts against client totals. The retained +`acknowledged-manifest.json` hashes the stream and source journals; the runner +checks the stream before and after both audits. Missing successful records fail +the case. No multi-million-command response list is held in memory. +Build manifests pin the collector, control scripts and celld adapter bytes. +Cases preserve the exact loaded runner source. Rebuild old artifact directories +into fresh directories before using the current JSONL runner; retain their +original runner and JSON-array journals as historical evidence. + Individual reports expose achieved rate, failed delivery gates, window costs and confirmed commands per selected root. They cannot establish the complete [proposal qualification](../../docs/write-performance-proposal.md): that also @@ -111,3 +129,18 @@ Cellule stability reports fit the debt and oldest-publication age over the last three one-minute segments. A positive slope fails; missing, late, reset, or fenced observations cannot pass. Native log frontiers remain observations and grant no durability or collection authority. + +For a machine-readable comparison, write an external JSON file with `baseline`, +`candidate` and `celld` lists of case directories, then run: + +```sh +python3 scripts/perf/compare.py /absolute/external/matrix.json \ + --output /absolute/external/comparison.json +``` + +The comparison requires identical offered point sets, driver binaries, pinned +images, fixture bytes, loaded runner, Docker host, resources and point +distributions, including hot reads and preceding write offsets. Historical +cases lacking host/runner or fixture records are explicitly marked unverified +for those identities. Its read ratios describe each matched point; +they do not establish the baseline's highest qualified read capacity. diff --git a/scripts/perf/build.py b/scripts/perf/build.py index b3e6e421..cdd80b5d 100644 --- a/scripts/perf/build.py +++ b/scripts/perf/build.py @@ -104,6 +104,8 @@ def build(args): 'crates/cellule-runtime/src/follower/records/append.rs', 'crates/cellule-runtime/src/follower/tests/append.rs', 'crates/cellule-axum/examples/capacity/mod.rs', + 'crates/cellule-axum/examples/capacity/arrivals/mod.rs', + 'crates/cellule-axum/examples/capacity/arrivals/tests.rs', 'crates/cellule-axum/examples/sql.rs', 'crates/cellule-axum/examples/sql_metrics/mod.rs', 'crates/cellule-axum/examples/sql_metrics/capture.rs', @@ -115,6 +117,7 @@ def build(args): 'crates/cellule-axum/examples/fleet/transport.rs', ] for name in names: + (source / name).parent.mkdir(parents=True, exist_ok=True) shutil.copy2(ROOT / name, source / name) overlay[name] = sha(source / name) manifest = {str(path.relative_to(source)): sha(path) for path in source.rglob('*') if path.is_file()} @@ -132,7 +135,7 @@ def build(args): sort_keys=True).encode()).hexdigest() target = cache / cache_key target.mkdir(exist_ok=True) - for name in ('control.py', 'wait-store-ready.py'): + for name in ('control.py', 'wait-store-ready.py', 'observations.py'): shutil.copy2(ROOT / 'scripts/perf' / name, destination / name) shutil.copytree(ROOT / 'scripts/perf/celld', destination / 'celld-app') subprocess.run(['python3', str(ROOT / 'scripts/generate-capacity-tls.py'), str(destination / 'tls')], check=True) @@ -144,7 +147,8 @@ def build(args): for name in ('sql', 'http_capacity', 'http_audit'): shutil.copy2(target / 'release/examples' / name, destination / 'bin' / name) binaries = {name: {'path': f'bin/{name}', 'sha256': sha(destination / f'bin/{name}')} for name in ('sql', 'http_capacity', 'http_audit')} - data = {'schema_version': 2, 'build_source_sha256': cache_key, 'framework_revision': resolved, 'measurement_overlay': overlay, 'framework_source_manifest_sha256': sha(destination / 'framework-source.json'), 'adapted_source': {str(p.relative_to(source)): sha(p) for p in sorted(examples.rglob('*.rs'))}, 'images': {'rust': RUST, 'celld': CELLD, 'store': RUSTFS}, 'celld_revision': 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'workload': 'sql-ledger-96', 'binaries': binaries} + fixture_names = ('control.py', 'wait-store-ready.py', 'observations.py', 'celld-app/index.js', 'celld-app/wrangler.json') + data = {'schema_version': 2, 'acknowledgement_format': 'jsonl-v1', 'fixture_sources': {name: sha(destination / name) for name in fixture_names}, 'build_source_sha256': cache_key, 'framework_revision': resolved, 'measurement_overlay': overlay, 'framework_source_manifest_sha256': sha(destination / 'framework-source.json'), 'adapted_source': {str(p.relative_to(source)): sha(p) for p in sorted(examples.rglob('*.rs'))}, 'images': {'rust': RUST, 'celld': CELLD, 'store': RUSTFS}, 'celld_revision': 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'workload': 'sql-ledger-96', 'binaries': binaries} (destination / 'build.json').write_text(json.dumps(data, sort_keys=True, indent=2) + '\n') print(json.dumps({'manifest': str(destination / 'build.json'), 'binaries': binaries})) if __name__ == '__main__': diff --git a/scripts/perf/compare.py b/scripts/perf/compare.py index 6f45bea8..70c13655 100644 --- a/scripts/perf/compare.py +++ b/scripts/perf/compare.py @@ -10,7 +10,7 @@ CONTRACT = ('durability', 'resident_cells', 'concurrency', 'queue_capacity', 'retained_budget_bytes', 'managed_disk_budget_bytes', 'value_bytes', 'owner_cpus', 'owner_memory_bytes', 'followers', 'profile', - 'provider_storage', 'seconds', 'warmup_seconds') + 'provider_storage', 'seconds', 'warmup_seconds', 'telemetry') def summarize(directory): @@ -20,6 +20,11 @@ def summarize(directory): report = case_report(directory) if hashlib.sha256(build_path.read_bytes()).hexdigest() != report['build_manifest_sha256']: raise ValueError('build manifest changed after the run: ' + str(directory)) + if build.get('acknowledgement_format') == 'jsonl-v1': + if not case.get('docker_host') or not case.get('runner_sha256'): + raise ValueError('bounded-audit case lacks host or runner provenance') + if hashlib.sha256((directory / 'runner.py').read_bytes()).hexdigest() != case['runner_sha256']: + raise ValueError('runner source changed after the run: ' + str(directory)) return case, build, report @@ -48,7 +53,7 @@ def compare(matrix): if set(matrix) != {'baseline', 'candidate', 'celld'}: raise ValueError('matrix needs baseline, candidate and celld case lists') rows = {} - contract = binaries = images = None + contract = binaries = images = fixtures = execution = None evidence = [] for role, directories in matrix.items(): if not directories: @@ -60,11 +65,18 @@ def compare(matrix): if case['system'] != ('celld' if role == 'celld' else 'cellule'): raise ValueError('wrong system for ' + role) identity = {key: case[key] for key in CONTRACT} + # Historical array-audit cases lack the complete execution record. + # Preserve them as evidence, but never label that provenance verified. + current = build.get('acknowledgement_format') == 'jsonl-v1' + environment = {'docker_host': case.get('docker_host'), + 'runner_sha256': case.get('runner_sha256')} if current else None driver = {key: build['binaries'][key]['sha256'] for key in ('http_capacity', 'http_audit')} if contract is None: - contract, binaries, images = identity, driver, build['images'] - if identity != contract or driver != binaries or build['images'] != images: + contract, binaries, images, fixtures = identity, driver, build['images'], build.get('fixture_sources') + execution = environment + if (identity != contract or driver != binaries or build['images'] != images + or build.get('fixture_sources') != fixtures or environment != execution): raise ValueError('workload, resources, client or image mismatch: ' + str(directory)) rows[role].append(report) evidence.append({'role': role, 'directory': str(directory), @@ -108,6 +120,8 @@ def compare(matrix): points['baseline'][key], points['candidate'][key]) if key[1] else None, 'arms': arms}) return {'schema_version': 1, 'workload': 'sql-ledger-96', 'contract': contract, + 'fixture_provenance_verified': fixtures is not None, + 'host_and_runner_provenance_verified': execution is not None, 'evidence': evidence, 'points': comparisons, 'qualification_pass': False, 'unverified': ['A/A sustainable-capacity search', 'read-only and mixed capacity guardrails', diff --git a/scripts/perf/http_audit.rs b/scripts/perf/http_audit.rs index daec4ef4..77391e0f 100644 --- a/scripts/perf/http_audit.rs +++ b/scripts/perf/http_audit.rs @@ -1,6 +1,12 @@ //! Comparison fixture: validate every acknowledged row and its exact stored retry. use serde::{Deserialize, Serialize}; -use std::{sync::Arc, time::Duration}; +use std::{ + io::{BufRead, Read}, + sync::Arc, + time::Duration, +}; +type AuditError = Box; +const MAX_OBSERVATION_BYTES: u64 = 64 * 1024; #[derive(Deserialize)] struct Config { address: String, @@ -34,95 +40,138 @@ struct Counts { changed_incarnations: u64, first_errors: Vec, } + +fn read_observation(reader: &mut impl BufRead) -> Result, AuditError> { + let mut bytes = Vec::new(); + let read = Read::take(reader, MAX_OBSERVATION_BYTES + 1).read_until(b'\n', &mut bytes)?; + if read == 0 { + return Ok(None); + } + if read as u64 > MAX_OBSERVATION_BYTES { + return Err("audit observation exceeds the fixture record bound".into()); + } + let observation: Observation = serde_json::from_slice(&bytes)?; + if !observation.request.is_object() { + return Err("audit requires the original request for every acknowledgement".into()); + } + Ok(Some(observation)) +} + +impl Counts { + fn merge(&mut self, counts: Self) { + self.checked += counts.checked; + self.retries_checked += counts.retries_checked; + self.errors += counts.errors; + self.changed_incarnations += counts.changed_incarnations; + self.first_errors.extend( + counts + .first_errors + .into_iter() + .take(4_usize.saturating_sub(self.first_errors.len())), + ); + } +} #[tokio::main(flavor = "multi_thread", worker_threads = 4)] async fn main() -> Result<(), Box> { let path = std::env::args() .nth(1) .ok_or("configuration path required")?; let config: Config = serde_json::from_slice(&std::fs::read(path)?)?; - let observations: Vec = - serde_json::from_slice(&std::fs::read(&config.observations)?)?; - if observations.iter().any(|observation| !observation.request.is_object()) { - return Err("audit requires the original request for every acknowledgement".into()); - } - let observations = Arc::new(observations); let config = Arc::new(config); + let (sender, mut observations) = tokio::sync::mpsc::channel(256); + let source = config.observations.clone(); + let reading = tokio::task::spawn_blocking(move || { + let mut reader = std::io::BufReader::new(std::fs::File::open(source)?); + while let Some(observation) = read_observation(&mut reader)? { + sender + .blocking_send(observation) + .map_err(|_| "audit consumer stopped")?; + } + Ok::<_, AuditError>(()) + }); let client = reqwest::Client::builder() .http1_only() .no_proxy() .timeout(Duration::from_secs(60)) .build()?; let mut tasks = tokio::task::JoinSet::new(); - for index in 0..128 { - let (observations, config, client) = (observations.clone(), config.clone(), client.clone()); + let mut total = Counts::default(); + while let Some(expected) = observations.recv().await { + if tasks.len() >= 128 + && let Some(counts) = tasks.join_next().await + { + total.merge(counts?); + } + let (config, client) = (config.clone(), client.clone()); tasks.spawn(async move { let mut counts = Counts::default(); - for expected in observations.iter().skip(index).step_by(128) { - let result = async { - let actual: Observation = client - .get(format!( - "http://{}/orders/{}", - config.address, expected.output.id - )) - .send() - .await? - .error_for_status()? - .json() - .await?; - if actual.output != expected.output - || actual.receipt.cell != expected.receipt.cell - || actual.receipt.commit_sequence < expected.receipt.commit_sequence - || (!config.cold - && actual.receipt.incarnation != expected.receipt.incarnation) - { - return Err(format!( - "row {} differs from acknowledged output or receipt", - expected.output.id - ) - .into()); - } - let retry: serde_json::Value = client - .post(format!("http://{}/orders", config.address)) - .json(&expected.request) - .send().await?.error_for_status()?.json().await?; - if retry["output"] != serde_json::to_value(&expected.output)? - || retry["receipt"]["cell"].as_str() != Some(expected.receipt.cell.as_str()) - || retry["receipt"]["commit_sequence"].as_u64() != Some(expected.receipt.commit_sequence) - || (!config.cold && retry["receipt"]["incarnation"].as_str() - != Some(expected.receipt.incarnation.as_str())) - { - return Err(format!("row {} stored retry differs", expected.output.id).into()); - } - Ok::<_, Box>( - actual.receipt.incarnation != expected.receipt.incarnation, + let result = async { + let actual: Observation = client + .get(format!( + "http://{}/orders/{}", + config.address, expected.output.id + )) + .send() + .await? + .error_for_status()? + .json() + .await?; + if actual.output != expected.output + || actual.receipt.cell != expected.receipt.cell + || actual.receipt.commit_sequence < expected.receipt.commit_sequence + || (!config.cold && actual.receipt.incarnation != expected.receipt.incarnation) + { + return Err(format!( + "row {} differs from acknowledged output or receipt", + expected.output.id ) + .into()); } - .await; - counts.checked += 1; - match result { - Ok(changed) => { - counts.changed_incarnations += u64::from(changed); - counts.retries_checked += 1; - } - Err(error) => { - counts.errors += 1; - if counts.first_errors.len() < 4 { - counts.first_errors.push(error.to_string()); - } + let retry: serde_json::Value = client + .post(format!("http://{}/orders", config.address)) + .json(&expected.request) + .send() + .await? + .error_for_status()? + .json() + .await?; + if retry["output"] != serde_json::to_value(&expected.output)? + || retry["receipt"]["cell"].as_str() != Some(expected.receipt.cell.as_str()) + || retry["receipt"]["commit_sequence"].as_u64() + != Some(expected.receipt.commit_sequence) + || (!config.cold + && retry["receipt"]["incarnation"].as_str() + != Some(expected.receipt.incarnation.as_str())) + { + return Err(format!("row {} stored retry differs", expected.output.id).into()); + } + Ok::<_, AuditError>(actual.receipt.incarnation != expected.receipt.incarnation) + } + .await; + counts.checked += 1; + match result { + Ok(changed) => { + counts.changed_incarnations += u64::from(changed); + counts.retries_checked += 1; + } + Err(error) => { + counts.errors += 1; + if counts.first_errors.len() < 4 { + counts.first_errors.push(error.to_string()); } } } counts }); } - let mut total = Counts::default(); while let Some(result) = tasks.join_next().await { - let counts = result?; - total.checked += counts.checked; - total.retries_checked += counts.retries_checked; - total.errors += counts.errors; - total.changed_incarnations += counts.changed_incarnations; - total.first_errors.extend(counts.first_errors); + total.merge(result?); + } + // The producer may fail after earlier records were checked. Such a partial + // audit must fail even when all dispatched GET/retry pairs succeeded. + reading.await??; + if total.checked == 0 { + return Err("empty acknowledgement audit".into()); } println!("{}", serde_json::to_string(&total)?); if total.errors > 0 { @@ -130,3 +179,36 @@ async fn main() -> Result<(), Box> { } Ok(()) } + +#[cfg(test)] +mod tests { + use super::*; + + const ROW: &[u8] = br#"{"output":{"id":1,"total_cents":2,"value":"p"},"receipt":{"cell":"c","incarnation":"i","commit_sequence":1},"request":{"id":1}}"#; + + #[test] + fn stream_reads_each_record_and_requires_original_requests() { + let bytes = [ROW, b"\n", ROW].concat(); + let mut reader = bytes.as_slice(); + assert!(read_observation(&mut reader).unwrap().is_some()); + assert!(read_observation(&mut reader).unwrap().is_some()); + assert!(read_observation(&mut reader).unwrap().is_none()); + let missing = String::from_utf8(ROW.to_vec()) + .unwrap() + .replace(r#""request":{"id":1}"#, r#""request":null"#); + assert!(read_observation(&mut missing.as_bytes()).is_err()); + } + + #[test] + fn malformed_or_oversize_record_fails_after_a_valid_prefix() { + for tail in [ + b"{broken".to_vec(), + vec![b'x'; MAX_OBSERVATION_BYTES as usize + 1], + ] { + let bytes = [ROW, b"\n", &tail].concat(); + let mut reader = bytes.as_slice(); + assert!(read_observation(&mut reader).unwrap().is_some()); + assert!(read_observation(&mut reader).is_err()); + } + } +} diff --git a/scripts/perf/observations.py b/scripts/perf/observations.py new file mode 100644 index 00000000..ef1320f5 --- /dev/null +++ b/scripts/perf/observations.py @@ -0,0 +1,64 @@ +"""Stream every ACK to disk with a disk-backed uniqueness check.""" +import json +import hashlib +import sqlite3 +import tempfile +from contextlib import closing +from pathlib import Path + + +def collect(directory, extra): + directory = Path(directory) + initialize = directory / 'initialize' + expected_seeds = json.loads((initialize / 'config.json').read_text())['cells'] + groups = [(initialize, [initialize / 'setup.jsonl'], expected_seeds)] + for phase in sorted(directory.glob('write-*')): + if not phase.is_dir(): + continue + writes = json.loads((phase / 'summary.json').read_text())['writes'] + expected = writes['successes_including_drain'] + writes['warmup_attempts'] - writes['warmup_errors'] + groups.append((phase, sorted(phase.glob('client-*.jsonl')), expected)) + sources = [] + count = 0 + with tempfile.TemporaryDirectory(prefix='ack-index-', dir=directory) as scratch: + with closing(sqlite3.connect(Path(scratch) / 'ids.sqlite')) as index, index: + # This index is disposable audit work, never a durability proof. + # Keep its cache bounded; a multi-million-command run cannot retain + # every output and request identity in a Python dictionary. + index.execute('PRAGMA cache_size=-2048') + index.execute('CREATE TABLE seen(id INTEGER PRIMARY KEY)') + with (directory / 'acknowledged.jsonl').open('x') as output: + def append(observation): + nonlocal count + oid = observation['output']['id'] + if type(oid) is not int or not isinstance(observation['request'], dict): + raise ValueError('ACK requires integer id and original request') + try: + index.execute('INSERT INTO seen VALUES (?)', (oid,)) + except sqlite3.IntegrityError as error: + raise ValueError(f'nonunique acknowledged id {oid}') from error + output.write(json.dumps(observation, separators=(',', ':')) + '\n') + count += 1 + if count % 10000 == 0: + index.commit() + append(extra) + for phase, paths, expected in groups: + before = count + for path in paths: + digest = hashlib.sha256() + with path.open('rb') as source: + for line in source: + digest.update(line) + row = json.loads(line) + if row.get('error') is None and row.get('response') and row.get('request'): + append(dict(row['response'], request=row['request'])) + sources.append({'path': str(path.relative_to(directory)), 'sha256': digest.hexdigest()}) + if count - before != expected: + raise ValueError(f'ACK journal count differs from client totals in {phase.name}: ' + f'{count - before} != {expected}') + with (directory / 'acknowledged.jsonl').open('rb') as stream: + digest = hashlib.file_digest(stream, 'sha256').hexdigest() + (directory / 'acknowledged-manifest.json').write_text(json.dumps( + {'schema_version': 1, 'format': 'jsonl', 'checked_acks': count, 'sha256': digest, + 'sources': sources}, indent=2) + '\n') + return count diff --git a/scripts/perf/report.py b/scripts/perf/report.py index 43b80d20..ce4b029a 100644 --- a/scripts/perf/report.py +++ b/scripts/perf/report.py @@ -7,6 +7,21 @@ def read(path): return json.loads(path.read_text()) +def expected_acknowledgements(case, summary): + phases = list(summary.get('writes', [])) + phases.extend(summary.get('overload', {}).get(name) for name in ('overload', 'recovery')) + total = case['resident_cells'] + 1 # Every seed plus the contract mutation. + for phase in phases: + if phase is None: + continue + writes = phase['writes'] + successes, attempts, errors = (writes[key] for key in ( + 'successes_including_drain', 'warmup_attempts', 'warmup_errors')) + if any(type(value) is not int for value in (successes, attempts, errors)) or not 0 <= errors <= attempts or successes < 0: + raise ValueError('invalid acknowledgement counters') + total += successes + attempts - errors + return total + def subtract(before, after): if isinstance(after, dict): return {key: subtract(before[key], value) for key, value in after.items() if key in before and (isinstance(value, (dict, list)) or isinstance(value, (int, float)))} @@ -182,6 +197,12 @@ def case_report(directory): failures.append('explicit evidence exclusion: ' + json.dumps(read(exclusions), sort_keys=True)) if not summary['completed']: failures.append(summary.get('failure') or 'case incomplete') + try: + expected_acks = expected_acknowledgements(case, summary) + except (KeyError, ValueError): + expected_acks = None + if expected_acks is None or summary.get('acknowledged_rows') != expected_acks: + failures.append('acknowledgement journals do not reconcile with client totals') for label in ('warm_audit', 'cold_audit'): audit = summary.get(label, {}) expected = summary.get('acknowledged_rows') @@ -215,8 +236,8 @@ def case_report(directory): points.append({'offered_writes_per_second': config['write_rate'], 'offered_reads_per_second': config['read_rate'], 'successful_writes_per_second': writes['successful_requests_per_second'], 'successful_reads_per_second': data['reads']['successful_requests_per_second'], 'load_contract': load_contract, 'logical_value_bytes_per_second': writes['successful_requests_per_second'] * 96, 'writes': writes, 'reads': data['reads'], 'window_metrics': metrics, 'window_cost': cost_report(metrics, writes['successes_in_window']), 'publication_stability': stability, 'delivery_latency_audit_pass': not point_failures, 'failures': point_failures}) overload = summary.get('overload', {}) recovery = overload.get('recovery') - recovery_failures = delivery_failures(recovery, case['durability']) if recovery else ['recovery phase missing'] - return {'schema_version': 1, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} + recovery_failures = (delivery_failures(recovery, case['durability']) if recovery else ['recovery phase missing']) + failures + return {'schema_version': 1, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'expected_acknowledged_rows': expected_acks, 'acknowledgement_count_reconciled': expected_acks is not None and summary.get('acknowledged_rows') == expected_acks, 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} if __name__ == '__main__': parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('directory', type=Path) diff --git a/scripts/perf/run.py b/scripts/perf/run.py index d394f4ce..d017cd0f 100644 --- a/scripts/perf/run.py +++ b/scripts/perf/run.py @@ -1,5 +1,6 @@ import os, sys, json, subprocess, time, uuid, threading, traceback, hashlib, math, shutil, fcntl, re from pathlib import Path +from observations import collect as collect_observations BASE = Path(__file__).resolve().parent CTX = None RUST = 'rust:1.98.1-bookworm@sha256:93ce27a88655056a51dbdd8f5f2d7ddc071c7b0070fb288a37b5a285fc83971e' @@ -9,7 +10,9 @@ STORE = None ARGS = None MANIFEST = None -RUNNER_SHA256 = hashlib.sha256(Path(__file__).read_bytes()).hexdigest() +DOCKER_HOST = None +RUNNER_BYTES = Path(__file__).read_bytes() +RUNNER_SHA256 = hashlib.sha256(RUNNER_BYTES).hexdigest() CREDS = ['-e', 'AWS_ACCESS_KEY_ID=benchmark_access', '-e', 'AWS_SECRET_ACCESS_KEY=benchmark_secret_private', '-e', 'AWS_DEFAULT_REGION=us-east-1'] def docker(*args, check=True, timeout=700): @@ -200,23 +203,16 @@ def contract(directory): raise RuntimeError(f'contract failed: {result}') return dict(first['body'], request=body) -def collect_observations(directory, extra): - observations = {extra['output']['id']: extra} - for path in [directory / 'initialize' / 'setup.jsonl', *sorted(directory.glob('write-*/client-*.jsonl'))]: - for line in path.open(): - row = json.loads(line) - if row.get('error') is None and row.get('response') and row.get('request'): - response = row['response'] - oid = response['output']['id'] - if oid in observations: - raise RuntimeError(f'nonunique acknowledged id {oid}') - observations[oid] = dict(response, request=row['request']) - put(directory / 'acknowledged.json', list(observations.values())) - return len(observations) - def audit(directory, label, cold): + observations = directory / 'acknowledged.jsonl' + manifest = json.loads((directory / 'acknowledged-manifest.json').read_text()) + def verify_journal(): + with observations.open('rb') as stream: + if hashlib.file_digest(stream, 'sha256').hexdigest() != manifest['sha256']: + raise RuntimeError('acknowledgement journal changed after collection') + verify_journal() path = directory / f'{label}-config.json' - put(path, {'address': '127.0.0.1:8080', 'observations': f'/work/{directory.relative_to(BASE)}/acknowledged.json', 'cold': cold}) + put(path, {'address': '127.0.0.1:8080', 'observations': f'/work/{directory.relative_to(BASE)}/acknowledged.jsonl', 'cold': cold}) p = docker('exec', CONTROL, '/work/' + MANIFEST['binaries']['http_audit']['path'], f'/work/{path.relative_to(BASE)}', check=False, timeout=1200) (directory / f'{label}.log').write_text(p.stdout + p.stderr) try: @@ -225,6 +221,7 @@ def audit(directory, label, cold): result = {'error': p.stdout + p.stderr} result['exit_code'] = p.returncode put(directory / f'{label}.json', result) + verify_journal() if p.returncode: raise RuntimeError(f'{label} failed: {result}') return result @@ -233,8 +230,9 @@ def run_case(system, durability): label = f'{system}-{durability}-parity-' + ARGS.tag directory = BASE / label directory.mkdir(exist_ok=False) + (directory / 'runner.py').write_bytes(RUNNER_BYTES) prefix = label + '-' + uuid.uuid4().hex[:12] - put(directory / 'case.json', {'system': system, 'durability': durability, 'prefix': prefix, 'framework_commit': MANIFEST['framework_revision'] if system == 'cellule' else 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'resident_cells': 1000, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'candidate_binary': MANIFEST['binaries']['sql']['path'], 'celld_application': 'celld-app', 'retained_budget_bytes': int(ARGS.retained_bytes), 'managed_disk_budget_bytes': int(ARGS.disk_bytes), 'value_bytes': 96, 'owner_cpus': 8, 'owner_memory_bytes': 16 * 1024 ** 3, 'followers': 2 if durability == 'fleet' else 0, 'profile': 'shared-vm-sql-ledger-96', 'provider_storage': 'fresh Linux Docker volume', 'telemetry': ARGS.telemetry, 'warmup_seconds': ARGS.warmup, 'seconds': ARGS.seconds, 'runner_sha256': RUNNER_SHA256, 'diagnostic': ARGS.seconds < 300 or ARGS.warmup < 30 or ARGS.telemetry == 'off'}) + put(directory / 'case.json', {'system': system, 'durability': durability, 'prefix': prefix, 'framework_commit': MANIFEST['framework_revision'] if system == 'cellule' else 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'resident_cells': 1000, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'candidate_binary': MANIFEST['binaries']['sql']['path'], 'celld_application': 'celld-app', 'retained_budget_bytes': int(ARGS.retained_bytes), 'managed_disk_budget_bytes': int(ARGS.disk_bytes), 'value_bytes': 96, 'owner_cpus': 8, 'owner_memory_bytes': 16 * 1024 ** 3, 'followers': 2 if durability == 'fleet' else 0, 'profile': 'shared-vm-sql-ledger-96', 'provider_storage': 'fresh Linux Docker volume', 'telemetry': ARGS.telemetry, 'warmup_seconds': ARGS.warmup, 'seconds': ARGS.seconds, 'runner_sha256': RUNNER_SHA256, 'docker_host': DOCKER_HOST, 'diagnostic': ARGS.seconds < 300 or ARGS.warmup < 30 or ARGS.telemetry == 'off'}) names = [] stop = threading.Event() sampling = None @@ -450,9 +448,19 @@ def provision(): if ARGS.overload_capacity is not None and ARGS.overload_capacity <= 0: parser.error('overload capacity must be positive') DOCKER = ['docker', '--context', CTX] + info = json.loads(docker('info', '--format', '{{json .}}').stdout) + DOCKER_HOST = {key: info[key] for key in ('OperatingSystem', 'OSType', 'Architecture', + 'NCPU', 'MemTotal', 'KernelVersion', 'ServerVersion')} MANIFEST = json.loads((BASE / 'build.json').read_text()) if MANIFEST.get('schema_version') != 2 or not MANIFEST.get('build_source_sha256'): parser.error('rebuild with source-content-isolated caches before running this harness') + if MANIFEST.get('acknowledgement_format') != 'jsonl-v1': + parser.error('rebuild with the bounded JSONL acknowledgement auditor before running this harness') + for name, expected in MANIFEST['fixture_sources'].items(): + if hashlib.sha256((BASE / name).read_bytes()).hexdigest() != expected: + parser.error('fixture source differs from build manifest: ' + name) + if hashlib.sha256(Path(collect_observations.__code__.co_filename).read_bytes()).hexdigest() != MANIFEST['fixture_sources']['observations.py']: + parser.error('loaded acknowledgement collector differs from the exported fixture') for binary in MANIFEST['binaries'].values(): if hashlib.sha256((BASE / binary['path']).read_bytes()).hexdigest() != binary['sha256']: parser.error('binary hash differs from build manifest') diff --git a/scripts/tests/test_perf_compare.py b/scripts/tests/test_perf_compare.py index bf39f7ae..61d053f5 100644 --- a/scripts/tests/test_perf_compare.py +++ b/scripts/tests/test_perf_compare.py @@ -1,6 +1,9 @@ """A comparison must retain failed points and reject mismatched workloads.""" import copy +import hashlib +import json import sys +import tempfile import unittest from pathlib import Path from unittest.mock import patch @@ -64,7 +67,7 @@ def test_hot_distribution_and_preceding_load_must_match(self): self.compare() def test_different_client_resources_or_provider_cannot_compare(self): - for kind in ('client', 'resources', 'image'): + for kind in ('client', 'resources', 'image', 'fixtures', 'host', 'runner'): with self.subTest(kind=kind): self.setUp() case, build, _ = self.arms['candidate'] @@ -72,11 +75,43 @@ def test_different_client_resources_or_provider_cannot_compare(self): build['binaries']['http_capacity']['sha256'] = 'different' elif kind == 'resources': case['owner_cpus'] = 4 - else: + elif kind == 'image': build['images']['store'] = 'different' + elif kind == 'fixtures': + build['fixture_sources'] = {'observations.py': 'different'} + else: + for _, peer_build, _ in self.arms.values(): + peer_build['acknowledgement_format'] = 'jsonl-v1' + case['docker_host' if kind == 'host' else 'runner_sha256'] = 'different' with self.assertRaisesRegex(ValueError, 'mismatch'): self.compare() + def test_historical_missing_provenance_cannot_be_marked_verified(self): + result = self.compare() + self.assertFalse(result['fixture_provenance_verified']) + self.assertFalse(result['host_and_runner_provenance_verified']) + + def test_current_case_requires_unchanged_runner_and_host_record(self): + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + directory = root / 'case' + directory.mkdir() + (root / 'build.json').write_text(json.dumps({'acknowledgement_format': 'jsonl-v1'})) + runner = b'original runner\n' + (directory / 'runner.py').write_bytes(runner) + case = {'docker_host': {'NCPU': 8}, 'runner_sha256': hashlib.sha256(runner).hexdigest()} + report = {'build_manifest_sha256': hashlib.sha256((root / 'build.json').read_bytes()).hexdigest()} + with patch.object(comparison, 'case_report', return_value=report): + (directory / 'case.json').write_text(json.dumps(case)) + comparison.summarize(directory) + (directory / 'runner.py').write_bytes(b'changed runner\n') + with self.assertRaisesRegex(ValueError, 'runner source changed'): + comparison.summarize(directory) + del case['docker_host'] + (directory / 'case.json').write_text(json.dumps(case)) + with self.assertRaisesRegex(ValueError, 'host or runner provenance'): + comparison.summarize(directory) + def test_unmatched_offered_points_cannot_disappear_from_comparison(self): extra = copy.deepcopy(point()) extra['offered_reads_per_second'] = 20000 diff --git a/scripts/tests/test_perf_observations.py b/scripts/tests/test_perf_observations.py new file mode 100644 index 00000000..9b4f5d3c --- /dev/null +++ b/scripts/tests/test_perf_observations.py @@ -0,0 +1,73 @@ +"""Audits must retain every original request and fail duplicate ACK IDs.""" +import json +import sys +import tempfile +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parents[1] / 'perf')) +from observations import collect + + +def observation(oid): + return {'output': {'id': oid}, 'receipt': {'commit_sequence': oid}, + 'request': {'request_id': str(oid), 'id': oid}} + + +class ObservationTests(unittest.TestCase): + def write(self, directory, name, rows): + path = directory / name + path.parent.mkdir(exist_ok=True) + path.write_text(''.join(json.dumps(row) + '\n' for row in rows)) + if path.parent.name == 'initialize': + (path.parent / 'config.json').write_text(json.dumps({'cells': len(rows)})) + else: + successes = sum(row.get('error') is None for row in rows) + (path.parent / 'summary.json').write_text(json.dumps({'writes': { + 'successes_including_drain': successes, 'warmup_attempts': 0, 'warmup_errors': 0}})) + + def row(self, oid): + ack = observation(oid) + return {'response': {key: value for key, value in ack.items() if key != 'request'}, + 'request': ack['request'], 'error': None} + + def test_warmup_and_every_write_phase_enter_the_stream(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + self.write(directory, 'initialize/setup.jsonl', [self.row(1)]) + self.write(directory, 'write-100/client-0.jsonl', [self.row(2), {'error': 'refused'}]) + self.write(directory, 'write-overload/client-0.jsonl', [self.row(3)]) + self.write(directory, 'write-recovery/client-0.jsonl', [self.row(4)]) + self.assertEqual(collect(directory, observation(0)), 5) + rows = [json.loads(line) for line in (directory / 'acknowledged.jsonl').read_text().splitlines()] + self.assertEqual({row['output']['id'] for row in rows}, set(range(5))) + self.assertTrue(all(row['request']['request_id'] == str(row['output']['id']) for row in rows)) + self.assertFalse(list(directory.glob('ack-index-*'))) + manifest = json.loads((directory / 'acknowledged-manifest.json').read_text()) + self.assertEqual(manifest['checked_acks'], 5) + self.assertEqual(len(manifest['sources']), 4) + + def test_duplicate_acknowledgements_and_existing_output_refuse(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + self.write(directory, 'initialize/setup.jsonl', [self.row(1)]) + self.write(directory, 'write-100/client-0.jsonl', [self.row(1)]) + with self.assertRaisesRegex(ValueError, 'nonunique acknowledged id'): + collect(directory, observation(0)) + self.assertFalse(list(directory.glob('ack-index-*'))) + with self.assertRaises(FileExistsError): + collect(directory, observation(0)) + + def test_missing_success_journal_cannot_reduce_the_audited_denominator(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + self.write(directory, 'initialize/setup.jsonl', [self.row(1)]) + self.write(directory, 'write-100/client-0.jsonl', [self.row(2)]) + (directory / 'write-100/client-0.jsonl').unlink() + with self.assertRaisesRegex(ValueError, 'ACK journal count differs from client totals'): + collect(directory, observation(0)) + self.assertFalse((directory / 'acknowledged-manifest.json').exists()) + + +if __name__ == '__main__': + unittest.main() diff --git a/scripts/tests/test_perf_report.py b/scripts/tests/test_perf_report.py index df94a52e..97a7ad99 100644 --- a/scripts/tests/test_perf_report.py +++ b/scripts/tests/test_perf_report.py @@ -11,6 +11,17 @@ class DeliveryGateTests(unittest.TestCase): + def test_all_ack_count_includes_warmup_late_overload_and_recovery_successes(self): + def phase(steady, warm, errors): + return {'writes': {'successes_including_drain': steady, + 'warmup_attempts': warm, 'warmup_errors': errors}} + summary = {'writes': [phase(30000, 3000, 0)], + 'overload': {'overload': phase(900, 0, 0), 'recovery': phase(200, 0, 0)}} + self.assertEqual(REPORT.expected_acknowledgements({'resident_cells': 1000}, summary), 35101) + summary['writes'][0]['writes']['warmup_errors'] = 3001 + with self.assertRaisesRegex(ValueError, 'invalid acknowledgement counters'): + REPORT.expected_acknowledgements({'resident_cells': 1000}, summary) + def point(self): return {"driver_exit_code": 0, "writes": { "errors": 0, "warmup_errors": 0, "queue_dropped": 0, "warmup_queue_dropped": 0, From a1be4caa2f4480159d99b0de75161ca7dca31e5c Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 6 Oct 2026 22:03:24 -0700 Subject: [PATCH 4/8] Fix fenced empty-epoch drain and reject exhausted benchmark storage An empty object coverage queue requires no new writer CAS. Preserve pending-ticket fencing, contiguous rotation, member receipts and authority closure. The fenced-owner fleet test now drains in 1.3 seconds instead of waiting forever. Record provider byte and inode headroom throughout Docker runs. Retained volumes can exhaust inodes despite free bytes; refuse such startup and invalidate deficient measurements. --- .../docs/failover-and-followers.md | 6 +++ .../src/node/durability/object_coverage.rs | 7 ++++ .../src/node/durability/tests.rs | 36 ++++++++++++++++ scripts/perf/README.md | 6 +++ scripts/perf/report.py | 23 ++++++++++- scripts/perf/run.py | 41 ++++++++++++++++++- scripts/tests/test_perf_provider.py | 40 ++++++++++++++++++ scripts/tests/test_perf_report.py | 18 ++++++++ 8 files changed, 174 insertions(+), 3 deletions(-) create mode 100644 scripts/tests/test_perf_provider.py diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index 40add1d2..36949cf2 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -1763,6 +1763,12 @@ sequenceDiagram An optional successor hint may prefetch the authenticated root directory and hot pages, but it grants no authority. +An empty object-coverage queue needs no new coverage CAS, including after a +local lease fence. This no-op grants no proof: contiguous coverage, exact +member retirement, and the current authority close are still required. Pending +tickets continue to require the original live lease and remain retained after +fencing. + Clean node shutdown is broader: 1. Withdraw public admission and mark the session draining diff --git a/crates/cellule-runtime/src/node/durability/object_coverage.rs b/crates/cellule-runtime/src/node/durability/object_coverage.rs index 0ac52dc8..65614eea 100644 --- a/crates/cellule-runtime/src/node/durability/object_coverage.rs +++ b/crates/cellule-runtime/src/node/durability/object_coverage.rs @@ -37,6 +37,13 @@ impl ObjectCoverage { lease: &NodeLeaseGuard, tickets: &[CommitTicket], ) -> Result<()> { + // Drain may close an already-covered epoch after its local lease has + // fenced. An empty queue requires no new coverage CAS or proof. A late + // staged ticket still blocks begin_rotation's contiguous-coverage check; + // native member retirement and the fresh authority close remain required. + if tickets.is_empty() && self.pending()?.is_empty() { + return Ok(()); + } let _flushing = tokio::select! { guard = self.flushing.lock() => guard, () = lease.wait_fenced() => return Err(Error::Fenced), diff --git a/crates/cellule-runtime/src/node/durability/tests.rs b/crates/cellule-runtime/src/node/durability/tests.rs index 7c6b24e3..9fc02617 100644 --- a/crates/cellule-runtime/src/node/durability/tests.rs +++ b/crates/cellule-runtime/src/node/durability/tests.rs @@ -598,3 +598,39 @@ async fn shutdown_retries_after_object_coverage_and_is_idempotent() { durability.shutdown().await.unwrap(); durability.shutdown().await.unwrap(); } + +#[tokio::test] +async fn fenced_empty_coverage_flush_does_not_require_new_writer_authority() { + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + let authority = RecordingAuthority::default(); + let lease = lease(); + lease.fence(); + let coverage = ObjectCoverage::default(); + coverage + .flush(&gate, &authority, &lease, &[]) + .await + .unwrap(); + assert!(authority.0.lock().unwrap().coverage.is_empty()); + assert_eq!(gate.tiered_through(), 0); +} + +#[tokio::test] +async fn fenced_pending_coverage_flush_retains_tickets_without_advancing() { + let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); + let authority = RecordingAuthority::default(); + let lease = lease(); + let ticket = gate.issue(1).unwrap(); + let coverage = ObjectCoverage::default(); + coverage.stage(&gate, &[ticket]).unwrap(); + lease.fence(); + assert!(matches!( + coverage.flush(&gate, &authority, &lease, &[]).await, + Err(Error::Fenced) + )); + assert!(authority.0.lock().unwrap().coverage.is_empty()); + assert_eq!(gate.tiered_through(), 0); + assert!(matches!( + gate.begin_rotation(), + Err(Error::PendingPublication) + )); +} diff --git a/scripts/perf/README.md b/scripts/perf/README.md index 280d2d15..57c73562 100644 --- a/scripts/perf/README.md +++ b/scripts/perf/README.md @@ -20,6 +20,12 @@ data on the Linux filesystem, retained for investigation. Its name is recorded in `store-data/volume.json`; it is never a host filesystem bind mount. The build records every exported source digest, the adapted fixture source, immutable binary hashes and pinned image digests. Never reuse an artifact directory for a new build. +Retained volumes share the Docker filesystem's byte and inode budgets. Startup +requires at least 10 GiB and one million free inodes. The runner records provider +filesystem counts before, during and after each case; falling below 2 GiB or +200,000 free inodes invalidates measurement. Archive and hash old test volumes +before removing their exact containers/volumes, or expand the dedicated test +disk. A fresh volume alone does not reset the shared filesystem's capacity. Build caches are partitioned by every adapted source digest and the compiler image. Before measurement, the runner reads a persisted root from the provider and checks its codec version against the exported source; the small-root diff --git a/scripts/perf/report.py b/scripts/perf/report.py index ce4b029a..cfad4dec 100644 --- a/scripts/perf/report.py +++ b/scripts/perf/report.py @@ -7,6 +7,24 @@ def read(path): return json.loads(path.read_text()) +def provider_health(directory): + paths = [directory / f'{label}-provider-filesystem.json' for label in ('startup', 'before', 'after')] + if any(not path.exists() for path in paths): + return {'available': False, 'pass': False, 'reason': 'provider byte/inode observations missing'} + samples = [read(path) for path in paths] + journal = directory / 'docker-stats.jsonl' + if not journal.exists(): + return {'available': False, 'pass': False, 'reason': 'provider sampling journal missing'} + with journal.open() as stream: + for line in stream: + samples.append(json.loads(line).get('provider_filesystem', {'error': 'missing filesystem sample'})) + valid = all(sample.get('measurement_usable') is True for sample in samples) + valid = valid and samples[0].get('startup_usable') is True + return {'available': True, 'pass': valid, 'sample_count': len(samples), + 'minimum_free_bytes': min((sample.get('free_bytes', 0) for sample in samples)), + 'minimum_free_inodes': min((sample.get('inodes', {}).get('free', 0) for sample in samples)), + 'reason': None if valid else 'provider capacity or filesystem evidence failed'} + def expected_acknowledgements(case, summary): phases = list(summary.get('writes', [])) phases.extend(summary.get('overload', {}).get(name) for name in ('overload', 'recovery')) @@ -192,6 +210,9 @@ def stability_report(directory): def case_report(directory): case, summary = (read(directory / 'case.json'), read(directory / 'summary.json')) failures = [] + health = provider_health(directory) + if case.get('provider_filesystem_required') and not health['pass']: + failures.append('provider byte/inode health failed: ' + str(health['reason'])) exclusions = directory / 'qualification-exclusions.json' if exclusions.exists(): failures.append('explicit evidence exclusion: ' + json.dumps(read(exclusions), sort_keys=True)) @@ -237,7 +258,7 @@ def case_report(directory): overload = summary.get('overload', {}) recovery = overload.get('recovery') recovery_failures = (delivery_failures(recovery, case['durability']) if recovery else ['recovery phase missing']) + failures - return {'schema_version': 1, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'expected_acknowledged_rows': expected_acks, 'acknowledgement_count_reconciled': expected_acks is not None and summary.get('acknowledged_rows') == expected_acks, 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} + return {'schema_version': 1, 'provider_health': health, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'expected_acknowledged_rows': expected_acks, 'acknowledgement_count_reconciled': expected_acks is not None and summary.get('acknowledged_rows') == expected_acks, 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} if __name__ == '__main__': parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('directory', type=Path) diff --git a/scripts/perf/run.py b/scripts/perf/run.py index d017cd0f..98440488 100644 --- a/scripts/perf/run.py +++ b/scripts/perf/run.py @@ -15,6 +15,31 @@ RUNNER_SHA256 = hashlib.sha256(RUNNER_BYTES).hexdigest() CREDS = ['-e', 'AWS_ACCESS_KEY_ID=benchmark_access', '-e', 'AWS_SECRET_ACCESS_KEY=benchmark_secret_private', '-e', 'AWS_DEFAULT_REGION=us-east-1'] +def parse_provider_filesystem(output): + """The Linux volume and its inode budget are shared by retained cases.""" + lines = output.strip().splitlines() + if len(lines) != 4: + raise ValueError('provider df evidence has an unexpected shape') + result = {} + for label, line in zip(('blocks', 'inodes'), (lines[1], lines[3])): + fields = line.split() + if len(fields) != 6 or fields[-1] != '/data': + raise ValueError('provider df evidence has an unexpected mount') + total, used, free = map(int, fields[1:4]) + if total <= 0 or min(used, free) < 0 or used + free > total: + raise ValueError('provider df evidence has invalid counts') + result[label] = {'device': fields[0], 'total': total, 'used': used, 'free': free} + if result['blocks']['device'] != result['inodes']['device']: + raise ValueError('provider df device changed') + result['free_bytes'] = result['blocks']['free'] * 1024 + result['measurement_usable'] = result['free_bytes'] >= 2 * 1024 ** 3 and result['inodes']['free'] >= 200000 + result['startup_usable'] = result['free_bytes'] >= 10 * 1024 ** 3 and result['inodes']['free'] >= 1000000 + return result + +def provider_filesystem(): + result = parse_provider_filesystem(docker('exec', STORE, 'sh', '-c', 'df -Pk /data && df -Pi /data').stdout) + return dict(result, timestamp=time.time()) + def docker(*args, check=True, timeout=700): p = subprocess.run(DOCKER + list(map(str, args)), capture_output=True, text=True, timeout=timeout) if check and p.returncode: @@ -116,6 +141,10 @@ def snapshot(directory, label, names): if state['OOMKilled'] or not state['Running']: failures.append(f'{name}: {state}') put(directory / f'{label}-resources.json', data) + filesystem = provider_filesystem() + put(directory / f'{label}-provider-filesystem.json', filesystem) + if not filesystem['measurement_usable']: + failures.append('provider byte or inode capacity exhausted') if failures: raise RuntimeError('infrastructure failed: ' + '; '.join(failures)) @@ -128,7 +157,11 @@ def sampler(directory, stop): f.write(json.dumps({'timestamp': time.time(), 'error': str(error)}) + '\n') f.flush() continue - f.write(json.dumps({'timestamp': time.time(), 'stats': [json.loads(x) for x in p.stdout.splitlines() if x]}) + '\n') + try: + filesystem = provider_filesystem() + except Exception as error: + filesystem = {'error': str(error), 'measurement_usable': False} + f.write(json.dumps({'timestamp': time.time(), 'provider_filesystem': filesystem, 'stats': [json.loads(x) for x in p.stdout.splitlines() if x]}) + '\n') f.flush() stop.wait(3) @@ -232,7 +265,7 @@ def run_case(system, durability): directory.mkdir(exist_ok=False) (directory / 'runner.py').write_bytes(RUNNER_BYTES) prefix = label + '-' + uuid.uuid4().hex[:12] - put(directory / 'case.json', {'system': system, 'durability': durability, 'prefix': prefix, 'framework_commit': MANIFEST['framework_revision'] if system == 'cellule' else 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'resident_cells': 1000, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'candidate_binary': MANIFEST['binaries']['sql']['path'], 'celld_application': 'celld-app', 'retained_budget_bytes': int(ARGS.retained_bytes), 'managed_disk_budget_bytes': int(ARGS.disk_bytes), 'value_bytes': 96, 'owner_cpus': 8, 'owner_memory_bytes': 16 * 1024 ** 3, 'followers': 2 if durability == 'fleet' else 0, 'profile': 'shared-vm-sql-ledger-96', 'provider_storage': 'fresh Linux Docker volume', 'telemetry': ARGS.telemetry, 'warmup_seconds': ARGS.warmup, 'seconds': ARGS.seconds, 'runner_sha256': RUNNER_SHA256, 'docker_host': DOCKER_HOST, 'diagnostic': ARGS.seconds < 300 or ARGS.warmup < 30 or ARGS.telemetry == 'off'}) + put(directory / 'case.json', {'system': system, 'durability': durability, 'prefix': prefix, 'framework_commit': MANIFEST['framework_revision'] if system == 'cellule' else 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'resident_cells': 1000, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'candidate_binary': MANIFEST['binaries']['sql']['path'], 'celld_application': 'celld-app', 'retained_budget_bytes': int(ARGS.retained_bytes), 'managed_disk_budget_bytes': int(ARGS.disk_bytes), 'value_bytes': 96, 'owner_cpus': 8, 'owner_memory_bytes': 16 * 1024 ** 3, 'followers': 2 if durability == 'fleet' else 0, 'profile': 'shared-vm-sql-ledger-96', 'provider_storage': 'fresh Linux Docker volume', 'provider_filesystem_required': True, 'telemetry': ARGS.telemetry, 'warmup_seconds': ARGS.warmup, 'seconds': ARGS.seconds, 'runner_sha256': RUNNER_SHA256, 'docker_host': DOCKER_HOST, 'diagnostic': ARGS.seconds < 300 or ARGS.warmup < 30 or ARGS.telemetry == 'off'}) names = [] stop = threading.Event() sampling = None @@ -240,6 +273,10 @@ def run_case(system, durability): try: docker('restart', CONTROL) start_provider(directory / 'store-data') + filesystem = provider_filesystem() + put(directory / 'startup-provider-filesystem.json', filesystem) + if not filesystem['startup_usable']: + raise RuntimeError('provider lacks 10 GiB free bytes or one million free inodes; preserve old volumes before retrying') create_bucket() for attempt in range(60): try: diff --git a/scripts/tests/test_perf_provider.py b/scripts/tests/test_perf_provider.py new file mode 100644 index 00000000..2d227804 --- /dev/null +++ b/scripts/tests/test_perf_provider.py @@ -0,0 +1,40 @@ +"""A filesystem with free bytes can still fail through exhausted inodes.""" +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parents[1] / 'perf')) +import run as runner + + +class ProviderTests(unittest.TestCase): + def evidence(self, free=7864324, byte_blocks=164221168): + return ('Filesystem 1024-blocks Used Available Capacity Mounted on\n' + f'/dev/vdb1 205838168 32373416 {byte_blocks} 17% /data\n' + 'Filesystem Inodes IUsed IFree IUse% Mounted on\n' + f'/dev/vdb1 13107200 5242876 {free} 40% /data\n') + + def test_free_bytes_do_not_hide_inode_exhaustion(self): + result = runner.parse_provider_filesystem(self.evidence(free=5)) + self.assertGreater(result['free_bytes'], 100 * 1024**3) + self.assertFalse(result['measurement_usable']) + self.assertFalse(result['startup_usable']) + + def test_startup_requires_headroom_beyond_the_measurement_floor(self): + result = runner.parse_provider_filesystem(self.evidence(free=500000)) + self.assertTrue(result['measurement_usable']) + self.assertFalse(result['startup_usable']) + self.assertTrue(runner.parse_provider_filesystem(self.evidence())['startup_usable']) + self.assertFalse(runner.parse_provider_filesystem(self.evidence(byte_blocks=1))['measurement_usable']) + + def test_missing_changed_or_invalid_mount_evidence_fails(self): + for evidence in ('', self.evidence().replace('/data', '/other'), + self.evidence().replace('5242876', '-1'), + self.evidence().replace('/dev/vdb1 13107200', '/dev/vdc1 13107200')): + with self.subTest(evidence=evidence): + with self.assertRaises(ValueError): + runner.parse_provider_filesystem(evidence) + + +if __name__ == '__main__': + unittest.main() diff --git a/scripts/tests/test_perf_report.py b/scripts/tests/test_perf_report.py index 97a7ad99..a0663db1 100644 --- a/scripts/tests/test_perf_report.py +++ b/scripts/tests/test_perf_report.py @@ -11,6 +11,24 @@ class DeliveryGateTests(unittest.TestCase): + def test_provider_health_preserves_failed_and_missing_inode_samples(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + self.assertFalse(REPORT.provider_health(directory)['pass']) + sample = {'measurement_usable': True, 'startup_usable': True, + 'free_bytes': 100 * 1024**3, 'inodes': {'free': 7000000}} + for label in ('startup', 'before', 'after'): + (directory / f'{label}-provider-filesystem.json').write_text(json.dumps(sample)) + journal = directory / 'docker-stats.jsonl' + journal.write_text(json.dumps({'provider_filesystem': sample}) + '\n') + self.assertTrue(REPORT.provider_health(directory)['pass']) + journal.write_text(json.dumps({'provider_filesystem': dict(sample, measurement_usable=False, inodes={'free': 5})}) + '\n') + result = REPORT.provider_health(directory) + self.assertFalse(result['pass']) + self.assertEqual(result['minimum_free_inodes'], 5) + journal.write_text('{}\n') + self.assertFalse(REPORT.provider_health(directory)['pass']) + def test_all_ack_count_includes_warmup_late_overload_and_recovery_successes(self): def phase(steady, warm, errors): return {'writes': {'successes_including_drain': steady, From 96dfb5b02f4bc24359d54b012c220fee0b5de86f Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 6 Oct 2026 22:32:41 -0700 Subject: [PATCH 5/8] Use fixed arrays for inline directory hex decoding --- crates/cellule-ltx/src/replica/root.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/crates/cellule-ltx/src/replica/root.rs b/crates/cellule-ltx/src/replica/root.rs index 2f1018e8..42a1fb2c 100644 --- a/crates/cellule-ltx/src/replica/root.rs +++ b/crates/cellule-ltx/src/replica/root.rs @@ -236,7 +236,9 @@ fn parse_inline(value: &str) -> Result> { } value .as_bytes() - .chunks_exact(2) + .as_chunks::<2>() + .0 + .iter() .map(|pair| Ok((nibble(pair[0])? << 4) | nibble(pair[1])?)) .collect::>>() .map(Into::into) From aefadc123f079204eedefe2fb7a2598d25cc2361 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 6 Oct 2026 22:51:12 -0700 Subject: [PATCH 6/8] Finish fenced node cleanup without releasing Cell authority --- .../docs/failover-and-followers.md | 7 ++ crates/cellule-runtime/src/publication/mod.rs | 18 ++- .../cellule-runtime/src/publication/tests.rs | 117 ++++++++++++++++++ 3 files changed, 141 insertions(+), 1 deletion(-) diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index 36949cf2..9038f07f 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -1769,6 +1769,13 @@ member retirement, and the current authority close are still required. Pending tickets continue to require the original live lease and remain retained after fencing. +After a terminal node fence, Cell deactivation discards the local handle and +leaves the exact selected owner/root for takeover. It performs no release CAS +and grants no Idle or durability proof. A node fence racing the cleanup read +has the same outcome. A live node session still reconciles and releases an +individually fenced Cell through the existing authority path; storage and +control failures remain errors. + Clean node shutdown is broader: 1. Withdraw public admission and mark the session draining diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index 1b7d3f30..112b6c16 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -940,7 +940,14 @@ impl CellPublisher { /// /// Reloading first adopts an ambiguous publication or pure renewal. A new /// epoch or owner proves that recovery already belongs to another executor. + /// A fenced node session performs local cleanup only, retaining authority + /// for takeover without attempting a release CAS or granting a new proof. pub(crate) async fn release_after_fence(&mut self) -> Result<()> { + match self.check_node_lease() { + Ok(()) => {} + Err(Error::Fenced) => return Ok(()), + Err(error) => return Err(error), + } let expected = self.observed.value().clone(); let expected_owner = expected.owner.clone().ok_or(Error::Fenced)?; let expected_cell = expected.cell; @@ -978,7 +985,16 @@ impl CellPublisher { )); } self.observed = current; - self.release().await + match self.release().await { + Err(Error::Fenced) => match self.check_node_lease() { + // A terminal node fence may race the load or release. Local + // deactivation still finishes; authority stays with recovery. + Err(Error::Fenced) => Ok(()), + Err(error) => Err(error), + Ok(()) => Err(Error::Fenced), + }, + result => result, + } } fn check_node_lease(&self) -> Result<()> { diff --git a/crates/cellule-runtime/src/publication/tests.rs b/crates/cellule-runtime/src/publication/tests.rs index 2e18662d..a113cc58 100644 --- a/crates/cellule-runtime/src/publication/tests.rs +++ b/crates/cellule-runtime/src/publication/tests.rs @@ -80,6 +80,123 @@ async fn leased_preparation_checkpoint_checks_liveness_without_cell_cas() { #[derive(Default)] struct CoverageAuthority(std::sync::Mutex>); +#[tokio::test] +async fn fenced_node_cleanup_preserves_selected_authority_for_takeover() { + verify_node_cleanup(CleanupFence::BeforeRead).await; +} + +#[tokio::test] +async fn node_fence_during_cleanup_read_preserves_selected_authority() { + verify_node_cleanup(CleanupFence::DuringRead).await; +} + +#[tokio::test] +async fn live_node_cleanup_still_releases_selected_authority() { + verify_node_cleanup(CleanupFence::Live).await; +} + +#[derive(Clone, Copy)] +enum CleanupFence { + Live, + BeforeRead, + DuringRead, +} + +async fn verify_node_cleanup(fence: CleanupFence) { + let cell = CellId::from_bytes([101; 32]); + let incarnation = IncarnationId::from_bytes([102; 16]); + let lease = crate::NodeLeaseGuard::new(0, 60_000).unwrap(); + let armed = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let store = Store::new(Arc::new(InMemory::new())).with_read_request_observer({ + let armed = armed.clone(); + let lease = lease.clone(); + Arc::new(move |_| { + if armed.swap(false, std::sync::atomic::Ordering::SeqCst) { + lease.fence(); + } + }) + }); + let layout = CellStorageLayout::new(store, Path::from("fenced-node-cleanup"), [103; 16]); + let control = Control::initial( + cell, + incarnation, + Owner { + session: SessionId::from_bytes([104; 16]), + endpoint: "https://node.internal:8081".into(), + }, + Digest::from_bytes([105; 32]), + 1, + ) + .unwrap(); + layout + .store() + .create_strict( + &layout.control_path(cell.as_bytes()), + Bytes::from(control.encode().unwrap()), + ) + .await + .unwrap(); + let authority = CellAuthority::new(layout.clone()); + let observed = authority.load(cell).await.unwrap().unwrap(); + let replica = CellReplica::new( + layout.clone(), + *cell.as_bytes(), + *incarnation.as_bytes(), + Limits::default(), + ) + .unwrap(); + let directory = tempfile::tempdir().unwrap(); + let mut publisher = CellPublisher::new( + replica, + authority.clone(), + observed, + directory.path().to_owned(), + ) + .with_node_lease(lease.clone()); + let mut database = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); + database + .transaction(|tx| { + tx.execute_batch("CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES(42)") + }) + .unwrap(); + let cuts = database.capture_deferred().unwrap(); + let prepared = publisher.prepare_append(&cuts, 1, 1, None).await.unwrap(); + publisher.publish_prepared(&prepared, None).await.unwrap(); + let selected = layout + .store() + .get_with_etag(&layout.control_path(cell.as_bytes())) + .await + .unwrap(); + match fence { + CleanupFence::BeforeRead => lease.fence(), + CleanupFence::DuringRead => armed.store(true, std::sync::atomic::Ordering::SeqCst), + CleanupFence::Live => {} + } + // Node fencing is terminal. Local cleanup grants no Idle release, root + // selection or new proof; the exact owner/root stays available to takeover. + publisher.release_after_fence().await.unwrap(); + let after = layout + .store() + .get_with_etag(&layout.control_path(cell.as_bytes())) + .await + .unwrap(); + match fence { + CleanupFence::Live => { + assert_ne!(after.1, selected.1); + let current = authority.load(cell).await.unwrap().unwrap(); + assert_eq!(current.value().state, crate::control::ControlState::Idle); + assert!(current.value().owner.is_none()); + assert_eq!(current.value().ltx_root(), Some(prepared.root())); + } + CleanupFence::BeforeRead | CleanupFence::DuringRead => { + assert_eq!(after, selected); + assert!(matches!(publisher.release().await, Err(Error::Fenced))); + assert!(matches!(publisher.renew().await, Err(Error::Fenced))); + } + } + database.close().unwrap(); +} + #[derive(Default)] struct CoverageTelemetry(std::sync::atomic::AtomicUsize); From 19927d16604d36d20b6c69d2bc4b4ae9bd74c620 Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 6 Oct 2026 22:59:53 -0700 Subject: [PATCH 7/8] Preserve original fenced drain errors in recovery fault fixtures --- .../tests/node/fleet_receivers/mod.rs | 43 +++++++++++++++++-- .../recovery_resume/history.rs | 6 +++ .../fleet_receivers/recovery_resume/mod.rs | 39 +++++++++++++++++ .../docs/failover-and-followers.md | 12 +++--- crates/cellule-runtime/src/publication/mod.rs | 18 +------- .../cellule-runtime/src/publication/tests.rs | 8 +++- 6 files changed, 99 insertions(+), 27 deletions(-) diff --git a/crates/cellule-host/tests/node/fleet_receivers/mod.rs b/crates/cellule-host/tests/node/fleet_receivers/mod.rs index f6e82df7..b04581d9 100644 --- a/crates/cellule-host/tests/node/fleet_receivers/mod.rs +++ b/crates/cellule-host/tests/node/fleet_receivers/mod.rs @@ -277,9 +277,34 @@ impl Movement { async fn shutdown(&self) { self.receiver.shutdown().await.unwrap(); - self.source.node.shutdown().await.unwrap(); - for node in [&self.receiver, &self.source.node] { - assert_eq!(node.state(), NodeState::Stopped); + let source_fenced = match self.source.node.shutdown().await { + Ok(()) => false, + Err(Error::Facility { + name: "cell-runtime-drain", + source, + }) if matches!(self.source.lease.check(), Err(Error::Fenced)) + && has_fenced_source(source.as_ref()) => + { + true + } + Err(error) => panic!("unexpected source shutdown failure: {error:?}"), + }; + // Failed-source fixtures deliberately fence the source lease. Cleanup + // may finish before recovery changes its owner record, retaining the + // original Fenced error as required by the runtime drain contract. + // No other facility failure or remaining native resource is accepted. + for (node, state) in [ + (&self.receiver, NodeState::Stopped), + ( + &self.source.node, + if source_fenced { + NodeState::Draining + } else { + NodeState::Stopped + }, + ), + ] { + assert_eq!(node.state(), state); let stats = node.stats(); assert_eq!(stats.active_cells(), 0); assert_eq!(stats.worker_jobs(), 0); @@ -291,6 +316,18 @@ impl Movement { } } +fn has_fenced_source(mut error: &(dyn std::error::Error + 'static)) -> bool { + loop { + if matches!(error.downcast_ref::(), Some(Error::Fenced)) { + return true; + } + match error.source() { + Some(source) => error = source, + None => return false, + } + } +} + async fn apply(node: &CellNode, action: FleetAction) -> Arc { tokio::time::timeout(Duration::from_secs(5), async { loop { diff --git a/crates/cellule-host/tests/node/fleet_receivers/recovery_resume/history.rs b/crates/cellule-host/tests/node/fleet_receivers/recovery_resume/history.rs index 5de945e1..03aa85f1 100644 --- a/crates/cellule-host/tests/node/fleet_receivers/recovery_resume/history.rs +++ b/crates/cellule-host/tests/node/fleet_receivers/recovery_resume/history.rs @@ -1,5 +1,11 @@ use super::*; +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_source_shutdown_preserves_its_fence_and_exact_recovery() { + let (movement, action, idle, receipt) = failed_with_source_drain(1, 42, false, true).await; + finish(movement, action, &idle, 42, false, receipt).await; +} + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn failed_source_missing_or_corrupt_history_blocks_idle_and_ordinary_winner() { for (corrupt, ordinary) in [(false, false), (true, false), (false, true), (true, true)] { diff --git a/crates/cellule-host/tests/node/fleet_receivers/recovery_resume/mod.rs b/crates/cellule-host/tests/node/fleet_receivers/recovery_resume/mod.rs index 70d97e09..63a26961 100644 --- a/crates/cellule-host/tests/node/fleet_receivers/recovery_resume/mod.rs +++ b/crates/cellule-host/tests/node/fleet_receivers/recovery_resume/mod.rs @@ -18,6 +18,15 @@ async fn failed( writes: usize, value: i64, overlay: bool, +) -> (Movement, FleetAction, Control, Acknowledged) { + failed_with_source_drain(writes, value, overlay, false).await +} + +async fn failed_with_source_drain( + writes: usize, + value: i64, + overlay: bool, + drain_source: bool, ) -> (Movement, FleetAction, Control, Acknowledged) { let movement = Movement::new(128 << 20).await; let now = clock(); @@ -47,6 +56,36 @@ async fn failed( } else { movement.start_recovery(false).await; } + if drain_source { + let before = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let error = movement.source.node.shutdown().await.unwrap_err(); + assert!(matches!( + error, + Error::Facility { + name: "cell-runtime-drain", + .. + } + )); + assert!(has_fenced_source(&error)); + assert_eq!(movement.source.node.state(), NodeState::Draining); + assert_eq!( + movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value(), + before.value() + ); + } if writes == 0 { movement .source diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index 9038f07f..f24c958e 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -1769,12 +1769,12 @@ member retirement, and the current authority close are still required. Pending tickets continue to require the original live lease and remain retained after fencing. -After a terminal node fence, Cell deactivation discards the local handle and -leaves the exact selected owner/root for takeover. It performs no release CAS -and grants no Idle or durability proof. A node fence racing the cleanup read -has the same outcome. A live node session still reconciles and releases an -individually fenced Cell through the existing authority path; storage and -control failures remain errors. +Cell deactivation can close every native handle while preserving a fenced +release failure. Shutdown retains that original error even if recovery later +changes the owner record. Resource closure does not prove an Idle release or +successful session withdrawal; the host remains draining when a required +facility fails. A fault test that deliberately fences its source must verify +the typed original failure and empty resource ledgers alongside exact recovery. Clean node shutdown is broader: diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index 112b6c16..1b7d3f30 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -940,14 +940,7 @@ impl CellPublisher { /// /// Reloading first adopts an ambiguous publication or pure renewal. A new /// epoch or owner proves that recovery already belongs to another executor. - /// A fenced node session performs local cleanup only, retaining authority - /// for takeover without attempting a release CAS or granting a new proof. pub(crate) async fn release_after_fence(&mut self) -> Result<()> { - match self.check_node_lease() { - Ok(()) => {} - Err(Error::Fenced) => return Ok(()), - Err(error) => return Err(error), - } let expected = self.observed.value().clone(); let expected_owner = expected.owner.clone().ok_or(Error::Fenced)?; let expected_cell = expected.cell; @@ -985,16 +978,7 @@ impl CellPublisher { )); } self.observed = current; - match self.release().await { - Err(Error::Fenced) => match self.check_node_lease() { - // A terminal node fence may race the load or release. Local - // deactivation still finishes; authority stays with recovery. - Err(Error::Fenced) => Ok(()), - Err(error) => Err(error), - Ok(()) => Err(Error::Fenced), - }, - result => result, - } + self.release().await } fn check_node_lease(&self) -> Result<()> { diff --git a/crates/cellule-runtime/src/publication/tests.rs b/crates/cellule-runtime/src/publication/tests.rs index a113cc58..e06a4426 100644 --- a/crates/cellule-runtime/src/publication/tests.rs +++ b/crates/cellule-runtime/src/publication/tests.rs @@ -174,7 +174,13 @@ async fn verify_node_cleanup(fence: CleanupFence) { } // Node fencing is terminal. Local cleanup grants no Idle release, root // selection or new proof; the exact owner/root stays available to takeover. - publisher.release_after_fence().await.unwrap(); + let result = publisher.release_after_fence().await; + match fence { + CleanupFence::Live => result.unwrap(), + CleanupFence::BeforeRead | CleanupFence::DuringRead => { + assert!(matches!(result, Err(Error::Fenced))); + } + } let after = layout .store() .get_with_etag(&layout.control_path(cell.as_bytes())) From a9e743ea148eb4e76b30f2b0c62f42dbce04f8bd Mon Sep 17 00:00:00 2001 From: forhappy Date: Tue, 6 Oct 2026 23:27:06 -0700 Subject: [PATCH 8/8] Retain final write comparison and verify provider lifecycle MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Record the final matched bucket point, all-ACK cold recovery, and remaining M0–M5 gaps. Capture provider state through cold audits and preserve OOM or missing evidence as qualification failures. --- docs/write-performance-delivery.md | 221 +++++++++++++++++++++++++++- scripts/perf/README.md | 5 + scripts/perf/report.py | 19 ++- scripts/perf/run.py | 19 ++- scripts/tests/test_perf_provider.py | 15 ++ scripts/tests/test_perf_report.py | 25 ++++ 6 files changed, 297 insertions(+), 7 deletions(-) diff --git a/docs/write-performance-delivery.md b/docs/write-performance-delivery.md index 86ed56e8..ed4e7d3a 100644 --- a/docs/write-performance-delivery.md +++ b/docs/write-performance-delivery.md @@ -1,6 +1,7 @@ # Write performance implementation and verification -The implementation packs small publication dependencies and provides a pinned +This delivers an implementation slice and a quantified gap report, not the +completed M0–M5 plan. The implementation packs small publication dependencies and provides a pinned Docker comparison with exact retry and cold-state audits. **Celld write parity has not been established.** The [proposal](write-performance-proposal.md) remains the acceptance contract; completing tests or a load run does not pass its gates. @@ -38,7 +39,7 @@ collection paths. There is no legacy decoding or automatic migration. | M2 | Existing native grouping and per-Cell coalescing preserved | Shared node publication coordinator not implemented; 0.25 publication PUTs/command not achieved | | M3 | Fresh enrollment roles, peer phases and follower append measured | Signed append grants and durable grant fences not implemented; 0.05 enrollment GETs/command not achieved | | M4 | Existing authority-pinned Cell roots remain the object proof | Bundle coverage proof, transfer and collection protocol not implemented | -| M5 | One matched five-minute Fleet point and all-ACK warm/cold audits completed | Three repetitions, read/failure/overload matrix and absolute/relative parity unverified | +| M5 | Matched Fleet/read points and target stress with exact ACK audits delivered | Three repetitions, read/failure/overload matrix and absolute/relative parity unverified | ## Evidence @@ -94,6 +95,181 @@ three PUTs/command before shared data, node coverage or compaction. M3's measure 2.347 enrollment GETs/command is 46.9 times its 0.05 budget. Meeting those gates requires the specified coalescing/proof/grant protocols and their failure tests. +For the deterministic uniform schedule at 2,000 bucket writes/s, a Cell receives +one command every 500 ms. Waiting for its next command would already exceed the +200-ms bucket p99 target. The existing three-PUT per-Cell selection floor cannot +reach the proposed 0.05 PUT/command budget through cross-Cell data bundling alone. +This is a cost/latency bound under that schedule, not a measured capacity. M4 +needs a complete authority-pinned range proof before bucket responses can use +shared coverage; an uploaded bundle or live node epoch alone cannot supply it. + +### Fleet target stress and immediate step-down + +The rebuilt driver offered 15,000 writes/s for 300 seconds after 30 seconds +warmup. The candidate completed **476.80 writes/s inside the window**, with +3,224.5-ms scheduled p99, 61 request errors and 4,356,649 measured queue drops. +It generated every one of the 4.5 million scheduled offers. This is an overloaded +completion rate, not qualified capacity. All 190,175 successful acknowledgements +from setup, warmup, stress and both later phases passed warm and bucket-only cold +GET/exact-retry audits. Original fleet drain took 24.615 seconds. + +The window selected 28,468 command roots covering 145,103 logical commits: +5.097 commits/root. All provider PUT successes/acknowledged command fell to +1.000; fresh owner/receiver enrollment GETs/command were 0.083. These are +time-window ratios with maintenance included, not complete cohort accounting +of trailing work. Neither M2's 0.25 nor M3's 0.05 budget passed. + +Publication mean was 4,046 ms and dirty admission mean 5,409 ms. Worker round +trip averaged 9.28 ms and capture 0.65 ms. Follower worker calls averaged +47.9–48.4 ms, while their data-sync barriers averaged 0.0011–0.0013 ms on tmpfs. +The worker timing includes the native call; it does not identify its internal +CPU or filesystem costs. At window end, oldest unpublished work was 224.7 seconds +old and 76,435 node sequences awaited contiguous object coverage. The final +three-minute debt slope was +19,406.5 bytes/s and age slope +903.4 ms/s. + +Immediately after stress the harness offered 150 writes/s for 60 seconds, then +50 writes/s for 30 seconds, each without fresh warmup. Both had zero request +errors, queue drops or unissued offers, but scheduled p99 remained 471.6 and +495.0 ms. The 50/s phase failed latency recovery. The nominal reference of +100 writes/s was not qualified capacity; this exercises the phase/audit path, +not M5's required overload at 1.5 times a qualified reference. + +Celld's matching stress case failed through lease watchdog self-fencing: +both followers and the owner exited with code 3, without container OOM. +Its mixed healthy/failed window completed 627.14 writes/s with 7,071.1-ms p99, +435 request errors and 4,311,172 drops. The service was unavailable for its warm +audit of 337,803 acknowledgements; cold audit did not run. This establishes +availability and qualification failure, **not acknowledged-state loss or a +clean Fleet capacity comparison**. All failed journals, node logs and the +provider volume remain in the external artifacts. + +Latest main's retained attempt completed 121.84 writes/s with 3,683,013 request +errors and 780,436 queue drops, then zero successful step-down writes. Its +54,771 warm checks failed and drain exceeded 120 seconds. **This attempt is +excluded from framework performance attribution:** the shared provider ran out +of inodes by the end of the case, and the original harness had no inode samples. +The following bucket owner received S3 `InternalError: Disk full`; Linux reported +only five free inodes despite 44 GiB of free bytes. Main's window recorded 205 +transient PUT outcomes, including seven node-authority writes. The exact timing +and contribution of exhaustion are unknown. These failures cannot establish a +candidate availability advantage or a clean three-arm stress comparison. +The exclusion is retained in the machine-readable evidence, rather than deleting +the failed attempt. + +The first candidate and celld bucket attempts produced no throughput sample: +the former failed startup against the exhausted provider, and the latter could +not restart its control container. The dedicated Docker disk was expanded from +80 to 200 GiB without changing CPU/memory ceilings or tmpfs node state. The +runner now checks byte and inode headroom before, during and after each case; +missing or exhausted required observations fail qualification. One diagnostic +volume was archived with a SHA-256 manifest; other retained volumes remain. + +New artifact identities are `implementation-m1-bounded-audit` and +`implementation-main-bounded-audit`. Their SQL binary hashes remain identical +to the respective earlier builds; both use client `417f07b0424d` and auditor +`6595c24b0be2`. Fixture, runner and Docker host identities are recorded and must +match. Do not combine these results with the older client's paired point. + +### Fresh bucket target after storage reset + +With the drain-fixed candidate and monitored provider headroom, 2,000 offered +bucket writes/s for 300 seconds yielded 192.243 successful commands/s, zero +HTTP errors, 542,069 measured drops and 51,687 warmup drops. All 600,000 +measured offers were generated; scheduled p50/p95/p99 were 819.1/4,559.0/8,491.8 +ms. All 67,245 acknowledged commands passed warm and cold GET/exact-retry audits, +and original owner drain took 3.955 seconds. These overloaded completions do +not qualify sustainable capacity. + +All provider PUT successes/command were 4.241 and GET attempts/command 1.291; +the window materialized 1.038 commands per selected root. Worker round trip +averaged 1.539 ms and capture 0.535 ms, versus 460.5-ms publication and 648.1-ms +object-response wait. The 1,823 compaction observations averaged 5,707.4 ms. +These intervals overlap and must not be added into a latency breakdown. +Node-log debt was zero in bucket mode, but oldest-publication age had a positive +51.8-ms/s final trend. All 69 provider filesystem observations passed; minimum +free space was 166.3 GB and minimum free inodes 7,573,521. The earlier disk-full +failure is not an explanation for this target miss. + +Celld's matching window completed 905.077 writes/s with 1,075.3-ms scheduled +p99, three request errors, 328,218 measured drops and 24,986 warmup drops. All +600,000 offers were generated, and its warm audit passed all 307,794 acknowledged +commands and exact retries. At this overloaded point the candidate completed +21.24% of celld's rate, with 7.90 times its scheduled p99. These are point ratios, +not qualified-capacity ratios. Original owner drain took 9.761 seconds. During +cold audit the 2-GiB provider was OOM-killed at 05:27:57 UTC; the cold owner +then self-fenced after lease renewal failed. Only nine exact cold retries +completed. This is a provider availability failure: acknowledged-state loss is +unproven, and celld's cold durability is unverified in this case. Provider state +and the explicit cold-audit exclusion are retained in the report. + +Latest main's fresh matching window completed 104.710 writes/s, with zero HTTP +errors, 568,331 measured drops and 57,138 warmup drops. Scheduled p50/p95 were +1,367.0/7,131.3 ms; p99 exceeded the 10,000-ms histogram bound (954 overflow +samples), so no exact p99 or percentile ratio is reported. All 35,532 ACKs +passed warm and cold GET/exact-retry audits; original owner drain took 4.212 +seconds. All 70 filesystem observations passed, with at least 160.6 GB and +6,564,266 inodes free. + +The final framework build `19927d16` repeated the same offered point with the +same frozen runner, fixture, client, auditor and resources. It completed +163.033 writes/s, with zero HTTP errors, 550,834 measured drops and 54,079 warmup +drops; all 600,000 scheduled offers were generated. Scheduled p50/p95/p99 were +1,014.5/5,513.4/8,906.2 ms. All **56,088 ACKs** passed warm and cold GET/exact-retry +audits, and original owner drain took **9.656 seconds**. Cold startup took +32.710 seconds. Provider headroom passed all 69 filesystem observations, with +at least 158.2 GB and 6,136,282 inodes free. Oldest-publication age still grew +55.5 ms/s over the final three minute segments, so stability failed. + +| Fresh bucket target point | Latest main | Final packed candidate | celld | +| --- | ---: | ---: | ---: | +| Successful writes/s inside window | 104.710 | 163.033 | 905.077 | +| Scheduled p99, ms | >10,000 | 8,906.2 | 1,075.3 | +| HTTP errors / measured drops | 0 / 568,331 | 0 / 550,834 | 3 / 328,218 | +| Warm + cold ACKs checked by GET/exact retry | 35,532 each | 56,088 each | Warm 307,794; cold unavailable | +| Original owner drain, seconds | 4.212 | 9.656 | 9.761 | +| All successful storage API PUTs/command | 5.874 | 4.186 | Not instrumented | +| GET attempts including ranges/command | 1.592 | 1.218 | Not instrumented | +| Logical commands per selected root | 1.041 | 1.045 | Not instrumented | +| Delivery qualification | Fail | Fail | Fail | + +The final overloaded candidate completed 55.7% more writes than retained main, +with 28.7% fewer PUTs/command. It reached 18.01% of celld's measured rate and +8.28 times its scheduled p99. Logical value throughput was 15,651 bytes/s; +provider PUT bytes were 2.65 MB/s, including publication and coordination. +These are different numerators, not user payload versus wire-equivalent rates. +Its publication/object-response means were 536.7/766.4 ms; worker round trip +and capture averaged 1.852/0.644 ms. Compaction averaged 6,427.3 ms over 1,501 +observations. These overlapping intervals are not additive CPU service costs. + +The earlier candidate completed 83.6% more writes than main, with 27.8% fewer +PUTs/command and publication/object-response means of 460.5/648.1 ms. The final +candidate's completion rate was 15.2% lower than that sample. This is not an A/A +pair: the revision changed. Preserve both samples; three qualified A/A and +paired repetitions remain missing. Main's final publication-age trend was +negative; both candidate samples were positive. Neither the throughput ratios +nor successful audits establish sustainable capacity, stability improvement, +read guardrails or celld parity. The provider lifecycle checker now records +cold/final state as well as byte/inode headroom; an OOM or missing required +lifecycle observation fails future cases. + +The earlier immutable candidate is `a1be4caa`, artifact +`implementation-m1-drain-fixed`, SQL hash `7c774549a8f6`. The final candidate is +`19927d16`, artifact `implementation-m1-preserved-contract`, SQL hash +`0560f70c65fc`. The baseline artifact is +`implementation-main-drain-comparison`, SQL hash `85d30c0f93df`; client/auditor +hashes remain `417f07b0424d`/`6595c24b0be2`. Comparisons use the new identical +filesystem-monitoring runner `d70fad321f81`. Results from the former runner +remain separate. The final comparison is indexed by +`matched-verified-bucket-final-matrix.json` and +`matched-verified-bucket-final-report.json` outside the repository. + +The new lifecycle runner separately completed a five-second, one-write/s +diagnostic on the final binary: all 1,036 ACKs passed warm and cold GET/retry +audits, with all five required lifecycle observations healthy. The target +comparison deliberately retains its frozen runner; it does not acquire the +new runner's lifecycle evidence retroactively. A diagnostic is not capacity +qualification. + ### Read saturation evidence The same frozen clients offered 10,000 reads/s for five minutes after 30 seconds @@ -140,7 +316,7 @@ in the retained manifests. All candidate production bytes match PR 66's `a1c48fcd`; the final test expectation and documentation were edited after the binary export. A Git base revision alone does not identify an overlaid build. -The isolated all-feature workspace suite passed **1,837 tests** with **38 ignored** +The initial isolated all-feature workspace suite passed **1,839 tests** with **38 ignored** environment-dependent tests. Clippy and API documentation passed with warnings denied; format, boundaries, layout, Rust fences, links and SQL/peer contract checks passed. The first compaction run exposed an old range-GET expectation; @@ -148,10 +324,41 @@ the corrected test now requires zero range GETs and two complete pack GETs. That failure remains in the external evidence. The revised client/auditor passed 18 targeted Rust tests and warnings-denied Clippy; local LTX without replica features passed 58 tests including its doctest. -The Python comparison, report and collector checks passed 21 tests with warnings -treated as errors. These checks do not substitute for live overload, fault or +The Python comparison, report, collector and provider checks passed 28 tests +with warnings treated as errors. These checks do not substitute for live overload, fault or capacity qualification. +The native fleet process suite exposed an empty-epoch shutdown loop after a +local owner fence. A regression test failed before the fix; the existing +fenced-owner evacuation test stalled at canonical shutdown. Empty coverage +queues now perform no new writer CAS, while pending tickets still reject +fencing and contiguous rotation/member/authority checks remain required. The +original process case passed in 1.30 seconds after the fix, and all 14 runtime +durability tests passed. The complete isolated fleet process suite then passed +all **382 tests** in 589.07 seconds. The full workspace suite, all-target/all-feature +check, warnings-denied Clippy/API documentation, format, architecture/layout, +document and SQL/peer gates passed again after the fix. Process tests are counted +separately from workspace tests. + +Final framework revision `19927d16` passed **1,843 workspace tests**, with +**38 ignored**, and all **382 fleet process tests** in 602.78 seconds. Its +all-target/all-feature check, Rust 1.99 warnings-denied Clippy, and API docs +passed; local LTX without replica features passed **58 tests** including its +doctest. All 1,240 Rust/Cargo source files in the isolated verification snapshot +match the checkout. The Linux SQL binary is `0560f70c65fc`; the client and +auditor remain `417f07b0424d` and `6595c24b0be2`. + +The recovery fault fixture now accepts only the original typed `Fenced` error +from the deliberately fenced source, retaining the `Draining` state and every +zero-resource-ledger assertion. Healthy receivers must still stop successfully. +A new deterministic case shuts down that source before receiver takeover, +checks that the selected authority record is unchanged, then verifies exact +reconstruction and the stored retry result. Production shutdown preserves +authority-release failures; closing local handles grants no successful release. +Three publication assertions also cover lease expiry, node fencing, and live +release. The failed contract-changing cleanup experiment and its test failures +remain outside Git as excluded evidence; that behavior was reverted. + ## Reproduce and inspect Follow the [harness instructions](../scripts/perf/README.md). Build each revision @@ -161,6 +368,10 @@ and generated `report.json`. Content-based cache namespaces and persisted-root checks prevent stale codec reuse. Source contains reusable drivers and compact conclusions; caches, volumes, journals, metric windows, binaries and logs stay outside Git. Each retained provider volume has `store-data/volume.json`. +The external `delivery-evidence-index.json` hashes the final comparison, build, +case, runner, ACK stream, format smoke and verification manifest. The latter +pins the 1,240 Rust/Cargo sources and all final check logs. Failed and excluded +attempts retain their own scope; they are not overwritten by passing reruns. `scripts/perf/compare.py` accepts an external JSON matrix containing `baseline`, `candidate` and `celld` lists of case directories. It rejects different workload, diff --git a/scripts/perf/README.md b/scripts/perf/README.md index 57c73562..3ce4db89 100644 --- a/scripts/perf/README.md +++ b/scripts/perf/README.md @@ -26,6 +26,11 @@ filesystem counts before, during and after each case; falling below 2 GiB or 200,000 free inodes invalidates measurement. Archive and hash old test volumes before removing their exact containers/volumes, or expand the dedicated test disk. A fresh volume alone does not reset the shared filesystem's capacity. +Provider lifecycle snapshots cover startup, both resource boundaries, cold +startup and final cleanup before the intentional provider stop. OOM or exit +during cold recovery invalidates the audit even if the measured window had +healthy byte/inode headroom. Keep that failure separate from data-loss claims; +missing required lifecycle evidence cannot pass. Build caches are partitioned by every adapted source digest and the compiler image. Before measurement, the runner reads a persisted root from the provider and checks its codec version against the exported source; the small-root diff --git a/scripts/perf/report.py b/scripts/perf/report.py index cfad4dec..703a45b7 100644 --- a/scripts/perf/report.py +++ b/scripts/perf/report.py @@ -25,6 +25,20 @@ def provider_health(directory): 'minimum_free_inodes': min((sample.get('inodes', {}).get('free', 0) for sample in samples)), 'reason': None if valid else 'provider capacity or filesystem evidence failed'} +def provider_lifecycle(directory, required=False): + """Filesystem headroom does not prove the provider survived cold restore.""" + labels = ('startup', 'before', 'after', 'cold', 'final') + paths = [directory / f'{label}-provider-state.json' for label in labels] + states = {label: read(path) for label, path in zip(labels, paths) if path.exists()} + failed = [label for label, state in states.items() + if state.get('Running') is not True or state.get('OOMKilled') is not False + or any(state.get(flag) is not False for flag in ('Paused', 'Restarting', 'Dead'))] + missing = [label for label in labels if label not in states] + return {'available': not missing, 'pass': not failed and not missing, + 'required': required, 'failed_phases': failed, 'missing_phases': missing, + 'states': states, 'reason': 'provider exited, OOM or unavailable' if failed + else 'provider lifecycle observations missing' if missing else None} + def expected_acknowledgements(case, summary): phases = list(summary.get('writes', [])) phases.extend(summary.get('overload', {}).get(name) for name in ('overload', 'recovery')) @@ -213,6 +227,9 @@ def case_report(directory): health = provider_health(directory) if case.get('provider_filesystem_required') and not health['pass']: failures.append('provider byte/inode health failed: ' + str(health['reason'])) + lifecycle = provider_lifecycle(directory, case.get('provider_lifecycle_required', False)) + if lifecycle['failed_phases'] or (lifecycle['required'] and not lifecycle['pass']): + failures.append('provider lifecycle failed: ' + str(lifecycle['reason'])) exclusions = directory / 'qualification-exclusions.json' if exclusions.exists(): failures.append('explicit evidence exclusion: ' + json.dumps(read(exclusions), sort_keys=True)) @@ -258,7 +275,7 @@ def case_report(directory): overload = summary.get('overload', {}) recovery = overload.get('recovery') recovery_failures = (delivery_failures(recovery, case['durability']) if recovery else ['recovery phase missing']) + failures - return {'schema_version': 1, 'provider_health': health, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'expected_acknowledged_rows': expected_acks, 'acknowledgement_count_reconciled': expected_acks is not None and summary.get('acknowledged_rows') == expected_acks, 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} + return {'schema_version': 1, 'provider_health': health, 'provider_lifecycle': lifecycle, 'case': summary['case'], 'system': case['system'], 'durability': case['durability'], 'workload': 'sql-ledger-96', 'seconds': case['seconds'], 'warmup_seconds': case['warmup_seconds'], 'build_manifest_sha256': summary['build_manifest_sha256'], 'completed': summary['completed'], 'expected_acknowledged_rows': expected_acks, 'acknowledgement_count_reconciled': expected_acks is not None and summary.get('acknowledged_rows') == expected_acks, 'points': points, 'failures': failures, 'overload': {'available': bool(overload), 'reference_capacity': overload.get('reference_capacity'), 'reference_requires_paired_qualification': True, 'recovery_30_second_delivery_pass': bool(recovery) and not recovery_failures, 'recovery_failures': recovery_failures, 'drain_seconds': summary.get('drain_seconds'), 'safe_refusals_before_sql_verified': False}, 'qualification_pass': False, 'qualification_unverified': ['three paired repetitions', 'A/A variance', 'sustained debt slopes and age', 'read-only and mixed guardrails', 'safe overload refusals and qualified reference capacity']} if __name__ == '__main__': parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('directory', type=Path) diff --git a/scripts/perf/run.py b/scripts/perf/run.py index 98440488..f71e9cb2 100644 --- a/scripts/perf/run.py +++ b/scripts/perf/run.py @@ -49,6 +49,14 @@ def docker(*args, check=True, timeout=700): def put(path, obj): path.write_text(json.dumps(obj, indent=2) + '\n') +def record_provider_state(directory, label): + state = json.loads(docker('inspect', STORE).stdout)[0]['State'] + put(directory / f'{label}-provider-state.json', state) + if state.get('Running') is not True or state.get('OOMKilled') is not False or any( + state.get(flag) is not False for flag in ('Paused', 'Restarting', 'Dead')): + raise RuntimeError('provider lifecycle failed: ' + json.dumps(state, sort_keys=True)) + return state + def http(method, path, body=None, port=8080): args = ['exec', CONTROL, 'python3', '/work/control.py', method, f'http://127.0.0.1:{port}{path}'] if body is not None: @@ -141,6 +149,7 @@ def snapshot(directory, label, names): if state['OOMKilled'] or not state['Running']: failures.append(f'{name}: {state}') put(directory / f'{label}-resources.json', data) + record_provider_state(directory, label) filesystem = provider_filesystem() put(directory / f'{label}-provider-filesystem.json', filesystem) if not filesystem['measurement_usable']: @@ -265,7 +274,7 @@ def run_case(system, durability): directory.mkdir(exist_ok=False) (directory / 'runner.py').write_bytes(RUNNER_BYTES) prefix = label + '-' + uuid.uuid4().hex[:12] - put(directory / 'case.json', {'system': system, 'durability': durability, 'prefix': prefix, 'framework_commit': MANIFEST['framework_revision'] if system == 'cellule' else 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'resident_cells': 1000, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'candidate_binary': MANIFEST['binaries']['sql']['path'], 'celld_application': 'celld-app', 'retained_budget_bytes': int(ARGS.retained_bytes), 'managed_disk_budget_bytes': int(ARGS.disk_bytes), 'value_bytes': 96, 'owner_cpus': 8, 'owner_memory_bytes': 16 * 1024 ** 3, 'followers': 2 if durability == 'fleet' else 0, 'profile': 'shared-vm-sql-ledger-96', 'provider_storage': 'fresh Linux Docker volume', 'provider_filesystem_required': True, 'telemetry': ARGS.telemetry, 'warmup_seconds': ARGS.warmup, 'seconds': ARGS.seconds, 'runner_sha256': RUNNER_SHA256, 'docker_host': DOCKER_HOST, 'diagnostic': ARGS.seconds < 300 or ARGS.warmup < 30 or ARGS.telemetry == 'off'}) + put(directory / 'case.json', {'system': system, 'durability': durability, 'prefix': prefix, 'framework_commit': MANIFEST['framework_revision'] if system == 'cellule' else 'f2bf648663a610eefde71f3547ad61e9b896b1f0', 'resident_cells': 1000, 'concurrency': int(ARGS.concurrency), 'queue_capacity': int(ARGS.queue_capacity), 'candidate_binary': MANIFEST['binaries']['sql']['path'], 'celld_application': 'celld-app', 'retained_budget_bytes': int(ARGS.retained_bytes), 'managed_disk_budget_bytes': int(ARGS.disk_bytes), 'value_bytes': 96, 'owner_cpus': 8, 'owner_memory_bytes': 16 * 1024 ** 3, 'followers': 2 if durability == 'fleet' else 0, 'profile': 'shared-vm-sql-ledger-96', 'provider_storage': 'fresh Linux Docker volume', 'provider_filesystem_required': True, 'provider_lifecycle_required': True, 'telemetry': ARGS.telemetry, 'warmup_seconds': ARGS.warmup, 'seconds': ARGS.seconds, 'runner_sha256': RUNNER_SHA256, 'docker_host': DOCKER_HOST, 'diagnostic': ARGS.seconds < 300 or ARGS.warmup < 30 or ARGS.telemetry == 'off'}) names = [] stop = threading.Event() sampling = None @@ -273,6 +282,7 @@ def run_case(system, durability): try: docker('restart', CONTROL) start_provider(directory / 'store-data') + record_provider_state(directory, 'startup') filesystem = provider_filesystem() put(directory / 'startup-provider-filesystem.json', filesystem) if not filesystem['startup_usable']: @@ -378,6 +388,7 @@ def run_case(system, durability): owner = f'comparison-{label}-cold' names.append(owner) summary['cold_startup_seconds'] = start_node(system, 'bucket', 0, prefix, owner) + record_provider_state(directory, 'cold') summary['cold_audit'] = audit(directory, 'cold-audit', True) saved = json.loads((directory / 'contract.json').read_text()) retry = http('POST', '/orders', saved['request']) @@ -405,6 +416,12 @@ def run_case(system, durability): summary.setdefault('cleanup_failures', []).append(str(e)) (directory / f'{name}.log').write_text(logs(name)) docker('stop', '--time', '2', name, check=False) + try: + record_provider_state(directory, 'final') + except Exception as error: + summary['completed'] = False + summary.setdefault('infrastructure_failures', []).append(str(error)) + summary['failure'] = summary.get('failure') or str(error) (directory / 'store.log').write_text(logs(STORE)) put(directory / 'summary.json', summary) print(json.dumps({'case': label, 'completed': summary['completed'], 'failure': str(summary.get('failure'))[:500], 'write_rates': [x['writes']['successful_requests_per_second'] for x in summary['writes']], 'read_rates': [x.get('successful_requests_per_second') for x in summary['reads']]}), flush=True) diff --git a/scripts/tests/test_perf_provider.py b/scripts/tests/test_perf_provider.py index 2d227804..d668755c 100644 --- a/scripts/tests/test_perf_provider.py +++ b/scripts/tests/test_perf_provider.py @@ -1,13 +1,28 @@ """A filesystem with free bytes can still fail through exhausted inodes.""" import sys import unittest +import json +import tempfile from pathlib import Path +from types import SimpleNamespace +from unittest.mock import patch sys.path.insert(0, str(Path(__file__).parents[1] / 'perf')) import run as runner class ProviderTests(unittest.TestCase): + def test_provider_oom_snapshot_is_written_before_the_case_is_rejected(self): + state = {'Running': False, 'OOMKilled': True, 'ExitCode': 137, + 'Paused': False, 'Restarting': False, 'Dead': False} + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + with patch.object(runner, 'docker', return_value=SimpleNamespace( + stdout=json.dumps([{'State': state}]))): + with self.assertRaisesRegex(RuntimeError, 'provider lifecycle failed'): + runner.record_provider_state(directory, 'final') + self.assertEqual(json.loads((directory / 'final-provider-state.json').read_text()), state) + def evidence(self, free=7864324, byte_blocks=164221168): return ('Filesystem 1024-blocks Used Available Capacity Mounted on\n' f'/dev/vdb1 205838168 32373416 {byte_blocks} 17% /data\n' diff --git a/scripts/tests/test_perf_report.py b/scripts/tests/test_perf_report.py index a0663db1..f177a933 100644 --- a/scripts/tests/test_perf_report.py +++ b/scripts/tests/test_perf_report.py @@ -11,6 +11,31 @@ class DeliveryGateTests(unittest.TestCase): + def test_provider_oom_after_measurement_is_retained_as_a_cold_failure(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + healthy = {'Running': True, 'OOMKilled': False, 'Paused': False, + 'Restarting': False, 'Dead': False} + for label in ('startup', 'before', 'after', 'cold', 'final'): + (directory / f'{label}-provider-state.json').write_text(json.dumps(healthy)) + self.assertTrue(REPORT.provider_lifecycle(directory, True)['pass']) + (directory / 'final-provider-state.json').write_text(json.dumps( + dict(healthy, Running=False, OOMKilled=True, ExitCode=137))) + result = REPORT.provider_lifecycle(directory, True) + self.assertFalse(result['pass']) + self.assertEqual(result['failed_phases'], ['final']) + self.assertTrue(result['states']['final']['OOMKilled']) + + def test_missing_or_incomplete_lifecycle_evidence_cannot_pass(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) + result = REPORT.provider_lifecycle(directory, True) + self.assertFalse(result['available']) + self.assertFalse(result['pass']) + self.assertIn('cold', result['missing_phases']) + (directory / 'final-provider-state.json').write_text('{"Running": true}') + self.assertEqual(REPORT.provider_lifecycle(directory)['failed_phases'], ['final']) + def test_provider_health_preserves_failed_and_missing_inode_samples(self): with tempfile.TemporaryDirectory() as temporary: directory = Path(temporary)